agora inbox for pgsql-hackers@postgresql.orghelp / color / mirror / Atom feed
pglife and post-beta commits 510+ messages / 3 participants [nested] [flat]
* pglife and post-beta commits @ 2018-06-19 22:18 Bruce Momjian <bruce@momjian.us> 0 siblings, 0 replies; 510+ messages in thread From: Bruce Momjian @ 2018-06-19 22:18 UTC (permalink / raw) To: pgsql-hackers FYI, you can now see all the commits since the most recent beta in pglife (http://pglife.momjian.us/). It is the plus sign link after "11 beta". The link currently is: https://git.postgresql.org/gitweb/?p=postgresql.git;a=shortlog;h=refs/heads/master;hp=586e4e6df5b85d... -- Bruce Momjian <bruce@momjian.us> http://momjian.us EnterpriseDB http://enterprisedb.com + As you are, so once was I. As I am, so you will be. + + Ancient Roman grave inscription + ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v8 3/7] Row pattern recognition patch (planner). @ 2023-09-25 05:01 Tatsuo Ishii <ishii@postgresql.org> 0 siblings, 0 replies; 510+ messages in thread From: Tatsuo Ishii @ 2023-09-25 05:01 UTC (permalink / raw) --- src/backend/optimizer/plan/createplan.c | 23 ++++++++++++++++++----- src/backend/optimizer/plan/setrefs.c | 23 +++++++++++++++++++++++ src/include/nodes/plannodes.h | 15 +++++++++++++++ 3 files changed, 56 insertions(+), 5 deletions(-) diff --git a/src/backend/optimizer/plan/createplan.c b/src/backend/optimizer/plan/createplan.c index 34ca6d4ac2..469fcd156b 100644 --- a/src/backend/optimizer/plan/createplan.c +++ b/src/backend/optimizer/plan/createplan.c @@ -286,9 +286,10 @@ static WindowAgg *make_windowagg(List *tlist, Index winref, int ordNumCols, AttrNumber *ordColIdx, Oid *ordOperators, Oid *ordCollations, int frameOptions, Node *startOffset, Node *endOffset, Oid startInRangeFunc, Oid endInRangeFunc, - Oid inRangeColl, bool inRangeAsc, bool inRangeNullsFirst, - List *runCondition, List *qual, bool topWindow, - Plan *lefttree); + Oid inRangeColl, bool inRangeAsc, bool inRangeNullsFirst, List *runCondition, + RPSkipTo rpSkipTo, List *patternVariable, List *patternRegexp, List *defineClause, + List *defineInitial, + List *qual, bool topWindow, Plan *lefttree); static Group *make_group(List *tlist, List *qual, int numGroupCols, AttrNumber *grpColIdx, Oid *grpOperators, Oid *grpCollations, Plan *lefttree); @@ -2698,6 +2699,11 @@ create_windowagg_plan(PlannerInfo *root, WindowAggPath *best_path) wc->inRangeAsc, wc->inRangeNullsFirst, wc->runCondition, + wc->rpSkipTo, + wc->patternVariable, + wc->patternRegexp, + wc->defineClause, + wc->defineInitial, best_path->qual, best_path->topwindow, subplan); @@ -6601,8 +6607,10 @@ make_windowagg(List *tlist, Index winref, int ordNumCols, AttrNumber *ordColIdx, Oid *ordOperators, Oid *ordCollations, int frameOptions, Node *startOffset, Node *endOffset, Oid startInRangeFunc, Oid endInRangeFunc, - Oid inRangeColl, bool inRangeAsc, bool inRangeNullsFirst, - List *runCondition, List *qual, bool topWindow, Plan *lefttree) + Oid inRangeColl, bool inRangeAsc, bool inRangeNullsFirst, List *runCondition, + RPSkipTo rpSkipTo, List *patternVariable, List *patternRegexp, List *defineClause, + List *defineInitial, + List *qual, bool topWindow, Plan *lefttree) { WindowAgg *node = makeNode(WindowAgg); Plan *plan = &node->plan; @@ -6628,6 +6636,11 @@ make_windowagg(List *tlist, Index winref, node->inRangeAsc = inRangeAsc; node->inRangeNullsFirst = inRangeNullsFirst; node->topWindow = topWindow; + node->rpSkipTo = rpSkipTo, + node->patternVariable = patternVariable; + node->patternRegexp = patternRegexp; + node->defineClause = defineClause; + node->defineInitial = defineInitial; plan->targetlist = tlist; plan->lefttree = lefttree; diff --git a/src/backend/optimizer/plan/setrefs.c b/src/backend/optimizer/plan/setrefs.c index 5700bfb5cd..648bf5a425 100644 --- a/src/backend/optimizer/plan/setrefs.c +++ b/src/backend/optimizer/plan/setrefs.c @@ -2456,6 +2456,29 @@ set_upper_references(PlannerInfo *root, Plan *plan, int rtoffset) NRM_EQUAL, NUM_EXEC_QUAL(plan)); + /* + * Modifies an expression tree in each DEFINE clause so that all Var + * nodes reference outputs of a subplan. + */ + if (IsA(plan, WindowAgg)) + { + WindowAgg *wplan = (WindowAgg *) plan; + + foreach(l, wplan->defineClause) + { + TargetEntry *tle = (TargetEntry *) lfirst(l); + + tle->expr = (Expr *) + fix_upper_expr(root, + (Node *) tle->expr, + subplan_itlist, + OUTER_VAR, + rtoffset, + NRM_EQUAL, + NUM_EXEC_QUAL(plan)); + } + } + pfree(subplan_itlist); } diff --git a/src/include/nodes/plannodes.h b/src/include/nodes/plannodes.h index 1b787fe031..afa27bb45a 100644 --- a/src/include/nodes/plannodes.h +++ b/src/include/nodes/plannodes.h @@ -1096,6 +1096,21 @@ typedef struct WindowAgg /* nulls sort first for in_range tests? */ bool inRangeNullsFirst; + /* Row Pattern Recognition AFTER MACH SKIP clause */ + RPSkipTo rpSkipTo; /* Row Pattern Skip To type */ + + /* Row Pattern PATTERN variable name (list of String) */ + List *patternVariable; + + /* Row Pattern RPATTERN regular expression quantifier ('+' or ''. list of String) */ + List *patternRegexp; + + /* Row Pattern DEFINE clause (list of TargetEntry) */ + List *defineClause; + + /* Row Pattern DEFINE variable initial names (list of String) */ + List *defineInitial; + /* * false for all apart from the WindowAgg that's closest to the root of * the plan -- 2.25.1 ----Next_Part(Mon_Sep_25_14_26_30_2023_752)-- Content-Type: Text/X-Patch; charset=us-ascii Content-Transfer-Encoding: 7bit Content-Disposition: inline; filename="v8-0004-Row-pattern-recognition-patch-executor.patch" ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v2 3/3] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 150 ++++++++++++++++++++++++++-- 2 files changed, 142 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fe470de63f2..7b245daaf4e 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,46 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +/* + * Check if the buffer is already undergoing read AIO. If it is, assign the + * IO's wait reference to operation->io_wref, thereby allowing the caller to + * wait for that IO. + */ +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if ((buf_state & BM_IO_IN_PROGRESS) && !(buf_state & BM_VALID)) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if ((buf_state & BM_IO_IN_PROGRESS) && !(buf_state & BM_VALID)) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1729,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1748,66 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (unlikely(operation->foreign_io)) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + + /* + * Report and track this as a 'hit' for this backend, even + * though it must have started out as a miss in + * PinBufferForBlock(). The other backend (or ourselves, + * as part of a read started earlier) will track this as a + * 'read'. + */ + TRACE_POSTGRESQL_BUFFER_READ_DONE(operation->forknum, + operation->blocknum + operation->nblocks_done, + operation->smgr->smgr_rlocator.locator.spcOid, + operation->smgr->smgr_rlocator.locator.dbOid, + operation->smgr->smgr_rlocator.locator.relNumber, + operation->smgr->smgr_rlocator.backend, + true); + + if (BufferIsLocal(buffer)) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1893,43 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses of + * an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + * + * It's possible that another backend starts IO on the buffer between this + * check and the ReadBuffersCanStartIO(nowait = false) below. In that case + * we will synchronously wait for the IO below, but the window for that is + * small enough that it won't happen often enough to have a significant + * performance impact. + */ + if (ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + *nblocks_progress = 1; + operation->foreign_io = true; + + CheckReadBuffersOperation(operation, false); + + + return true; + } + + operation->foreign_io = false; + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1987,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O is in progress in another backend (it can't be + * this backend), we want to wait for the outcome: either done, or + * something went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { -- 2.48.1.76.g4e746b1a31.dirty --aknbaqiolq3mrqjk-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v2 3/3] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 150 ++++++++++++++++++++++++++-- 2 files changed, 142 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fe470de63f2..7b245daaf4e 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,46 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +/* + * Check if the buffer is already undergoing read AIO. If it is, assign the + * IO's wait reference to operation->io_wref, thereby allowing the caller to + * wait for that IO. + */ +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if ((buf_state & BM_IO_IN_PROGRESS) && !(buf_state & BM_VALID)) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if ((buf_state & BM_IO_IN_PROGRESS) && !(buf_state & BM_VALID)) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1729,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1748,66 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (unlikely(operation->foreign_io)) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + + /* + * Report and track this as a 'hit' for this backend, even + * though it must have started out as a miss in + * PinBufferForBlock(). The other backend (or ourselves, + * as part of a read started earlier) will track this as a + * 'read'. + */ + TRACE_POSTGRESQL_BUFFER_READ_DONE(operation->forknum, + operation->blocknum + operation->nblocks_done, + operation->smgr->smgr_rlocator.locator.spcOid, + operation->smgr->smgr_rlocator.locator.dbOid, + operation->smgr->smgr_rlocator.locator.relNumber, + operation->smgr->smgr_rlocator.backend, + true); + + if (BufferIsLocal(buffer)) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1893,43 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses of + * an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + * + * It's possible that another backend starts IO on the buffer between this + * check and the ReadBuffersCanStartIO(nowait = false) below. In that case + * we will synchronously wait for the IO below, but the window for that is + * small enough that it won't happen often enough to have a significant + * performance impact. + */ + if (ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + *nblocks_progress = 1; + operation->foreign_io = true; + + CheckReadBuffersOperation(operation, false); + + + return true; + } + + operation->foreign_io = false; + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1987,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O is in progress in another backend (it can't be + * this backend), we want to wait for the outcome: either done, or + * something went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { -- 2.48.1.76.g4e746b1a31.dirty --aknbaqiolq3mrqjk-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
* [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO @ 2025-08-15 15:01 Andres Freund <andres@anarazel.de> 0 siblings, 0 replies; 510+ messages in thread From: Andres Freund @ 2025-08-15 15:01 UTC (permalink / raw) Author: Reviewed-by: Discussion: https://postgr.es/m/ Backpatch: --- src/include/storage/bufmgr.h | 1 + src/backend/storage/buffer/bufmgr.c | 133 ++++++++++++++++++++++++++-- 2 files changed, 125 insertions(+), 9 deletions(-) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 41fdc1e7693..7ddb867bc99 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -137,6 +137,7 @@ struct ReadBuffersOperation int flags; int16 nblocks; int16 nblocks_done; + bool foreign_io; PgAioWaitRef io_wref; PgAioReturn io_return; }; diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index fd7e21d96d3..de755fd53ad 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1557,6 +1557,41 @@ ReadBuffersCanStartIOOnce(Buffer buffer, bool nowait) return StartBufferIO(GetBufferDescriptor(buffer - 1), true, nowait); } +static inline bool +ReadBuffersIOAlreadyInProgress(ReadBuffersOperation *operation, Buffer buffer) +{ + BufferDesc *desc; + uint32 buf_state; + PgAioWaitRef iow; + + pgaio_wref_clear(&iow); + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + + if (buf_state & BM_IO_IN_PROGRESS) + iow = desc->io_wref; + UnlockBufHdr(desc, buf_state); + } + + if (pgaio_wref_valid(&iow)) + { + operation->io_wref = iow; + return true; + } + + return false; +} + /* * Helper for AsyncReadBuffers that tries to get the buffer ready for IO. */ @@ -1689,7 +1724,7 @@ WaitReadBuffers(ReadBuffersOperation *operation) * * we first check if we already know the IO is complete. */ - if (aio_ret->result.status == PGAIO_RS_UNKNOWN && + if ((operation->foreign_io || aio_ret->result.status == PGAIO_RS_UNKNOWN) && !pgaio_wref_check_done(&operation->io_wref)) { instr_time io_start = pgstat_prepare_io_time(track_io_timing); @@ -1708,11 +1743,38 @@ WaitReadBuffers(ReadBuffersOperation *operation) Assert(pgaio_wref_check_done(&operation->io_wref)); } - /* - * We now are sure the IO completed. Check the results. This - * includes reporting on errors if there were any. - */ - ProcessReadBuffersResult(operation); + if (operation->foreign_io) + { + Buffer buffer = operation->buffers[operation->nblocks_done]; + BufferDesc *desc; + uint32 buf_state; + + if (BufferIsLocal(buffer)) + { + desc = GetLocalBufferDescriptor(-buffer - 1); + buf_state = pg_atomic_read_u32(&desc->state); + } + else + { + desc = GetBufferDescriptor(buffer - 1); + buf_state = LockBufHdr(desc); + UnlockBufHdr(desc, buf_state); + } + + if (buf_state & BM_VALID) + { + operation->nblocks_done += 1; + Assert(operation->nblocks_done <= operation->nblocks); + } + } + else + { + /* + * We now are sure the IO completed. Check the results. This + * includes reporting on errors if there were any. + */ + ProcessReadBuffersResult(operation); + } } /* @@ -1798,6 +1860,56 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) io_object = IOOBJECT_RELATION; } + /* + * If AIO is in progress, be it in this backend or another backend, we + * just associate the wait reference with the operation and wait in + * WaitReadBuffers(). This turns out to be important for performance in + * two workloads: + * + * 1) A read stream that has to read the same block multiple times within + * the readahead distance. This can happen e.g. for the table accesses + * of an index scan. + * + * 2) Concurrent scans by multiple backends on the same relation. + * + * If we were to synchronously wait for the in-progress IO, we'd not be + * able to keep enough I/O in flight. + * + * + * If we do find there is ongoing I/O for the buffer, we set up a 1-block + * ReadBuffersOperation that WaitReadBuffers then can wait on. + */ + if (1 && ReadBuffersIOAlreadyInProgress(operation, buffers[nblocks_done])) + { + /* FIXME: probably need to wait if io_method == sync? */ + + *nblocks_progress = 1; + did_start_io = true; + operation->foreign_io = true; + + if (0) + elog(LOG, "using foreign IO path"); + + /* FIXME: trace point */ + + /* + * FIXME: how should this be accounted for in stats? Account as a hit + * for now, quite likely *we* started this IO. + */ + if (persistence == RELPERSISTENCE_TEMP) + pgBufferUsage.local_blks_hit += 1; + else + pgBufferUsage.shared_blks_hit += 1; + + if (operation->rel) + pgstat_count_buffer_hit(operation->rel); + pgstat_count_io_op(io_object, io_context, IOOP_HIT, 1, 0); + if (VacuumCostActive) + VacuumCostBalance += VacuumCostPageHit; + + return true; + } + /* * If zero_damaged_pages is enabled, add the READ_BUFFERS_ZERO_ON_ERROR * flag. The reason for that is that, hopefully, zero_damaged_pages isn't @@ -1855,9 +1967,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) /* * Check if we can start IO on the first to-be-read buffer. * - * If an I/O is already in progress in another backend, we want to wait - * for the outcome: either done, or something went wrong and we will - * retry. + * If a synchronous I/O in progress in another backend (it can't be this + * backend), we want to wait for the outcome: either done, or something + * went wrong and we will retry. */ if (!ReadBuffersCanStartIO(buffers[nblocks_done], false)) { @@ -1970,6 +2082,7 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) if (VacuumCostActive) VacuumCostBalance += VacuumCostPageMiss * io_buffers_len; + operation->foreign_io = false; *nblocks_progress = io_buffers_len; did_start_io = true; } @@ -5986,6 +6099,8 @@ WaitIO(BufferDesc *buf) */ if (pgaio_wref_valid(&iow)) { + if (0) + elog(LOG, "foreign wait"); pgaio_wref_wait(&iow); /* -- 2.48.1.76.g4e746b1a31.dirty --j4imdtjxqt5tb6mo-- ^ permalink raw reply [nested|flat] 510+ messages in thread
end of thread, other threads:[~2025-08-15 15:01 UTC | newest] Thread overview: 510+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2018-06-19 22:18 pglife and post-beta commits Bruce Momjian <bruce@momjian.us> 2023-09-25 05:01 [PATCH v8 3/7] Row pattern recognition patch (planner). Tatsuo Ishii <ishii@postgresql.org> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v2 3/3] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v2 3/3] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de> 2025-08-15 15:01 [PATCH v1] bufmgr: aio: Prototype for not waiting for already-in-progress IO Andres Freund <andres@anarazel.de>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox