agora inbox for pgsql-hackers@postgresql.org
help / color / mirror / Atom feed[PATCH v1 1/2] Persist slot invalidations before publishing them
32+ messages / 9 participants
[nested] [flat]
* [PATCH v1 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by:
Discussion:
Backpatch-through: 14
---
src/backend/replication/slot.c | 140 +++++++++++++-----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 1 +
.../t/056_replslot_invalidation_durability.pl | 125 ++++++++++++++++
4 files changed, 235 insertions(+), 33 deletions(-)
56.6% src/backend/replication/
41.1% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..b5746eb6283 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotReleaseOnError(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,18 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Release a slot claimed internally for invalidation after an error.
+ */
+static void
+ReplicationSlotReleaseOnError(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1197,29 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+
+ Assert(MyReplicationSlot != NULL);
+ Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR,
+ NameStr(MyReplicationSlot->data.name));
+
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2047,9 +2098,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2108,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2159,8 +2196,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2206,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(
+ invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2426,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2539,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2519,7 +2565,9 @@ CreateSlotOnDisk(ReplicationSlot *slot)
* Shared functionality between saving and creating a replication slot.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2575,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,9 +2584,11 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
+
LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
/* silence valgrind :( */
@@ -2576,6 +2628,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2669,6 +2735,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 39ec8c4946d..a74b9c64a4a 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -64,6 +64,7 @@ tests += {
't/053_standby_login_event_trigger.pl',
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
+ 't/056_replslot_invalidation_durability.pl',
],
},
}
diff --git a/src/test/recovery/t/056_replslot_invalidation_durability.pl b/src/test/recovery/t/056_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/056_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
--
2.34.1
--3JG95Zw2Rwjv7Ebv
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment;
filename="v1-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
^ permalink raw reply [nested|flat] 32+ messages in thread
* [PATCH v2 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 140 +++++++++++++-----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 1 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++
4 files changed, 235 insertions(+), 33 deletions(-)
56.6% src/backend/replication/
41.1% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..b5746eb6283 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotReleaseOnError(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,18 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Release a slot claimed internally for invalidation after an error.
+ */
+static void
+ReplicationSlotReleaseOnError(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1197,29 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+
+ Assert(MyReplicationSlot != NULL);
+ Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR,
+ NameStr(MyReplicationSlot->data.name));
+
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2047,9 +2098,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2108,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2159,8 +2196,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2206,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(
+ invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2426,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2539,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2519,7 +2565,9 @@ CreateSlotOnDisk(ReplicationSlot *slot)
* Shared functionality between saving and creating a replication slot.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2575,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,9 +2584,11 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
+
LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
/* silence valgrind :( */
@@ -2576,6 +2628,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2669,6 +2735,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..9248a7390a6 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,7 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
--
2.34.1
--Cg2b3K2/P/2A21mh
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v2-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
Content-Transfer-Encoding: 8bit
^ permalink raw reply [nested|flat] 32+ messages in thread
* [PATCH v3 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 234 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 587 insertions(+), 49 deletions(-)
45.2% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..8f2b681b626 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,32 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The lock is released on success and left held for error cleanup
+ * otherwise.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2063,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2117,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2138,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2148,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2158,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2186,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2246,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2256,17 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,15 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock. On error, leave the lock held for error cleanup. The lock is
+ * released after successful shared-memory publication.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2628,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2638,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2659,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2686,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2714,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2736,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2753,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2770,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,6 +2797,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
--1qxyLN8W7y4rfpkV
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v3-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
Content-Transfer-Encoding: 8bit
^ permalink raw reply [nested|flat] 32+ messages in thread
* [PATCH v4 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 236 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 588 insertions(+), 50 deletions(-)
45.1% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..642babebeab 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,31 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The caller is responsible for releasing the lock.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2062,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2116,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2137,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2147,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2157,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2185,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2245,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2255,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
+ LWLockRelease(&s->io_in_progress_lock);
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2627,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2637,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2658,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2685,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2713,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2735,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2752,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2769,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2796,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
--eAb0l2fUAigRz87W
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v4-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
Content-Transfer-Encoding: 8bit
^ permalink raw reply [nested|flat] 32+ messages in thread
* [PATCH v5 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 236 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 588 insertions(+), 50 deletions(-)
45.1% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..642babebeab 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,31 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The caller is responsible for releasing the lock.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2062,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2116,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2137,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2147,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2157,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2185,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2245,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2255,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
+ LWLockRelease(&s->io_in_progress_lock);
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2627,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2637,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2658,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2685,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2713,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2735,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2752,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2769,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2796,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
--WoN85jNTt/56ixen
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v5-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
Content-Transfer-Encoding: 8bit
^ permalink raw reply [nested|flat] 32+ messages in thread
* [PATCH v6 1/2] Persist slot invalidations before publishing them
@ 2026-08-26 05:34 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 05:34 UTC (permalink / raw)
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A failed save now leaves both the shared slot and its disk image valid. On
ERROR, release ownership claimed for an inactive slot before releasing the I/O
lock while preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add tests using an injection point to cover failed saves, subsequent
checkpointing and immediate restart, and concurrent internal invalidators.
This changes neither the on disk slot format nor the ReplicationSlot shared
memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 229 +++++++++++----
src/include/replication/slot.h | 3 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 582 insertions(+), 50 deletions(-)
43.6% src/backend/replication/
54.9% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..ef54833e0f9 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,16 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +772,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +788,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +821,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +836,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -1168,7 +1184,46 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ *
+ * The caller must own the slot and hold its I/O lock. On success, the caller
+ * retains both. On ERROR, release slot ownership before the I/O lock and
+ * update inactive_since only if requested.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn,
+ bool update_inactive_since)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ PG_TRY();
+ {
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
+ }
+ PG_CATCH();
+ {
+ HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
+ ReplicationSlotReleaseInternal(update_inactive_since);
+ LWLockRelease(&slot->io_in_progress_lock);
+ PG_RE_THROW();
+ }
+ PG_END_TRY();
}
/*
@@ -2005,6 +2060,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2114,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2135,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2145,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2155,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2183,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2243,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,10 +2253,14 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED,
+ false);
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
+ LWLockRelease(&s->io_in_progress_lock);
ReportSlotInvalidation(invalidation_cause, false, active_pid,
slotname, restart_lsn,
@@ -2380,7 +2468,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2581,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2605,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2620,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2630,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2651,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2678,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2706,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2728,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2745,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2762,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2789,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..885d3af236c 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,9 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn,
+ bool update_inactive_since);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
--d0VzUwhPUKEsymmV
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v6-0002-Persist-synchronized-slot-invalidations-before-pu.patch"
Content-Transfer-Encoding: 8bit
^ permalink raw reply [nested|flat] 32+ messages in thread
* Persist slot invalidations before publishing them
@ 2026-08-26 13:49 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 3 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-26 13:49 UTC (permalink / raw)
To: pgsql-hackers@lists.postgresql.org; +Cc: Amit Kapila <amit.kapila16@gmail.com>
Hi hackers,
while reviewing [1], I hit an issue due to the fact that an inactive replication
slot is marked invalid in shared memory before its new state is persisted.
If ReplicationSlotSave() errors before replacing the state file, the slot is
invalid in shared memory but still valid on disk. That sounds problematic as the
resource horizon computations could stop accounting for the slot, remove required
WAL or rows, and then an immediate restart would restore the old valid slot image.
The same issue exists in synchronize_one_slot(): it copies the invalidation from
the remote slot into the local synchronized slot before saving it. In that case,
a save error also prevents a direct retry because the next synchronization sees
the local slot as already invalid and skips it.
The InvalidatePossiblyObsoleteSlot() ordering seems to come from c6550776394e.
4ae08cd5fd19 later made those invalidations persistent but kept the same ordering.
PFA a patch series to $SUBJECT.
It introduces ReplicationSlotPersistInvalidation(), which creates an invalidated
copy of the acquired slot and writes it while the shared slot remains valid.
That means that a concurrent slot saver either writes the old valid state before
the invalidation operation, or waits and snapshots the invalid state after it has
been published. If the invalidated image can not be written, both the shared and
on disk states remain valid.
This is the same kind of idea used of effective_catalog_xmin and in 3741f2a09d52.
The patch series is organized that way:
0001: persist InvalidatePossiblyObsoleteSlot() invalidations before publishing
them.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
It also adds an injection point and some tests.
0002: do the same for synchronize_one_slot().
It keeps the synchronized slot restart LSN when copying a WAL invalidation,
preserving the current behavior. It also recomputes the xmin and WAL horizons
once the invalidation is durable and visible.
It also adds some test.
Remarks:
1/ there is no new retry mechanism. A later checkpoint or synchronization
naturally retries because the shared slot remains valid after the error.
2/ the patches change neither the on disk slot format nor the ReplicationSlot
shared memory layout. It has been done that way to ease the back patching.
3/ I think 0001 should be backpatched down to 14. 14 and 15 would probably need
some adaptations though (I did not look in detail yet).
4/ 0002 should be backpatched down to 17, where failover slot synchronization was
introduced.
[1]: https://postgr.es/m/CALj2ACUi0LeqKzomuXNPFsekzuQ%2BXbWZ5RamnO6tDzG4i1-KLw%40mail.gmail.com
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v1-0001-Persist-slot-invalidations-before-publishing-them.patch (15.6K, ../../ao7u5I9OeIR72kGp@bdtpg/2-v1-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From c82ad716dffdb3491ad8e18f93a9d1d0b451139f Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v1 1/2] Persist slot invalidations before publishing them
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by:
Discussion:
Backpatch-through: 14
---
src/backend/replication/slot.c | 140 +++++++++++++-----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 1 +
.../t/056_replslot_invalidation_durability.pl | 125 ++++++++++++++++
4 files changed, 235 insertions(+), 33 deletions(-)
56.6% src/backend/replication/
41.1% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..b5746eb6283 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotReleaseOnError(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,18 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Release a slot claimed internally for invalidation after an error.
+ */
+static void
+ReplicationSlotReleaseOnError(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1197,29 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+
+ Assert(MyReplicationSlot != NULL);
+ Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR,
+ NameStr(MyReplicationSlot->data.name));
+
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2047,9 +2098,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2108,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2159,8 +2196,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2206,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(
+ invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2426,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2539,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2519,7 +2565,9 @@ CreateSlotOnDisk(ReplicationSlot *slot)
* Shared functionality between saving and creating a replication slot.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2575,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,9 +2584,11 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
+
LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
/* silence valgrind :( */
@@ -2576,6 +2628,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2669,6 +2735,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 39ec8c4946d..a74b9c64a4a 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -64,6 +64,7 @@ tests += {
't/053_standby_login_event_trigger.pl',
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
+ 't/056_replslot_invalidation_durability.pl',
],
},
}
diff --git a/src/test/recovery/t/056_replslot_invalidation_durability.pl b/src/test/recovery/t/056_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/056_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v1-0002-Persist-synchronized-slot-invalidations-before-pu.patch (6.8K, ../../ao7u5I9OeIR72kGp@bdtpg/3-v1-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 6fed1aa8d90009a6178beb3a8dc6198b982c3af4 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v1 2/2] Persist synchronized slot invalidations before
publishing them
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. It also prevents synchronization
from retrying the save directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, restart, retry,
and final durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by:
Discussion:
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 10 +-
src/backend/replication/slot.c | 2 +-
.../t/056_replslot_invalidation_durability.pl | 138 ++++++++++++++++++
3 files changed, 142 insertions(+), 8 deletions(-)
10.0% src/backend/replication/logical/
3.0% src/backend/replication/
86.8% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..51c19c60cf9 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,13 +829,9 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
-
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ ReplicationSlotPersistInvalidation(remote_slot->invalidated, false);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index b5746eb6283..02cec90a20b 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1211,7 +1211,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
char path[MAXPGPATH];
Assert(MyReplicationSlot != NULL);
- Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.persistency != RS_EPHEMERAL);
Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/056_replslot_invalidation_durability.pl b/src/test/recovery/t/056_replslot_invalidation_durability.pl
index 247d7b00dc1..a360d6ee8f9 100644
--- a/src/test/recovery/t/056_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/056_replslot_invalidation_durability.pl
@@ -122,4 +122,142 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'valid synchronized slot survives restart after failed save');
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-26 19:24 Miłosz Bieniek <bieniek.milosz@proton.me>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2 siblings, 1 reply; 32+ messages in thread
From: Miłosz Bieniek @ 2026-08-26 19:24 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org, Amit Kapila <amit.kapila16@gmail.com>
Hi Bertrand
> 3/ I think 0001 should be backpatched down to 14. 14 and 15 would probably need
> some adaptations though (I did not look in detail yet).
>
> 4/ 0002 should be backpatched down to 17, where failover slot synchronization was
> introduced.
I did a brief review and tried to test these changes locally.
I ran test for PG19 and master, and they passed, but couldn't
apply patches for PG18 and lower.
I'll try dig deeper into this patch later.
Kind regards,
Miłosz Bieniek
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-27 01:30 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: Miłosz Bieniek <bieniek.milosz@proton.me>
0 siblings, 0 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-27 01:30 UTC (permalink / raw)
To: Miłosz Bieniek <bieniek.milosz@proton.me>; +Cc: pgsql-hackers@lists.postgresql.org, Amit Kapila <amit.kapila16@gmail.com>
Hi Miłosz,
On Wed, Aug 26, 2026 at 07:24:07PM +0000, Miłosz Bieniek wrote:
> Hi Bertrand
>
> > 3/ I think 0001 should be backpatched down to 14. 14 and 15 would probably need
> > some adaptations though (I did not look in detail yet).
> >
> > 4/ 0002 should be backpatched down to 17, where failover slot synchronization was
> > introduced.
>
> I did a brief review and tried to test these changes locally.
Thanks for looking at it!
> I ran test for PG19 and master, and they passed, but couldn't
> apply patches for PG18 and lower.
Yeah, the back patch versions are not done. I think it's better to discuss the
issue first and agree on what the fix should be on master. Once done the
back patch versions will be prepared based on the agreement.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-27 08:06 Kyotaro Horiguchi <horikyota.ntt@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2 siblings, 1 reply; 32+ messages in thread
From: Kyotaro Horiguchi @ 2026-08-27 08:06 UTC (permalink / raw)
To: bertranddrouvot.pg@gmail.com; +Cc: pgsql-hackers@lists.postgresql.org; amit.kapila16@gmail.com
Hello,
At Wed, 26 Aug 2026 13:49:24 +0000, Bertrand Drouvot <bertranddrouvot.pg@gmail.com> wrote in
> If ReplicationSlotSave() errors before replacing the state file, the slot is
> invalid in shared memory but still valid on disk. That sounds problematic as the
> resource horizon computations could stop accounting for the slot, remove required
> WAL or rows, and then an immediate restart would restore the old valid slot image.
I've spent some time looking through the related discussions and
patches, and I think I now have a better understanding of the
problem. I have a question about the persistence mechanism.
For the InvalidatePossiblyObsoleteSlot() case, at least for
RS_INVAL_XID_AGE, if the server crashes after the slot is invalidated
but before the invalidation is persisted, it seems that the restored
slot would still satisfy the same XID-age condition and would
eventually be invalidated again by vacuum or checkpoint. Is the main
reason for making the invalidation durable here that we don't want to
leave the slot valid until that next opportunity?
If so, I'm a little uncomfortable with persisting a modified copy of
the normal slot state before that state has actually been published in
shared memory. It seems to make the state transition somewhat harder
to follow, since the slot state file no longer necessarily represents
the current slot state.
Would it be simpler to persist the invalidation separately? For
example, we could write the invalidation cause to a small file such as
pg_replslot/<slotname>/invalidated and make it durable before
publishing the invalidation in shared memory. On restart, that file
would cause the slot to be restored as invalidated with the recorded
cause. This would keep the normal slot state file as a representation
of the actual slot state, and would also naturally avoid the race with
concurrent slot saves.
Regards,
--
Kyotaro Horiguchi
NTT Open Source Software Center
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-27 10:26 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-27 10:26 UTC (permalink / raw)
To: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org; amit.kapila16@gmail.com
Hi Horiguchi-san,
On Thu, Aug 27, 2026 at 05:06:47PM +0900, Kyotaro Horiguchi wrote:
> Hello,
>
> At Wed, 26 Aug 2026 13:49:24 +0000, Bertrand Drouvot <bertranddrouvot.pg@gmail.com> wrote in
> > If ReplicationSlotSave() errors before replacing the state file, the slot is
> > invalid in shared memory but still valid on disk. That sounds problematic as the
> > resource horizon computations could stop accounting for the slot, remove required
> > WAL or rows, and then an immediate restart would restore the old valid slot image.
>
> I've spent some time looking through the related discussions and
> patches,
Thanks for looking at it!
> For the InvalidatePossiblyObsoleteSlot() case, at least for
> RS_INVAL_XID_AGE, if the server crashes after the slot is invalidated
> but before the invalidation is persisted, it seems that the restored
> slot would still satisfy the same XID-age condition and would
> eventually be invalidated again by vacuum or checkpoint. Is the main
> reason for making the invalidation durable here that we don't want to
> leave the slot valid until that next opportunity?
Yeah, for XID age the condition should still hold after restart. On a primary,
the end of recovery checkpoint should detect it before connections are accepted.
On a hot standby, however, connections can be accepted before the next successful
restartpoint, so a restored slot could be used while valid although rows it
needed may already have been removed.
Also, other causes are not necessarily rediscovered immediately. For example,
inactive_since is reset at startup, so an idle timeout invalidation would not
be detected again until the timeout has elapsed again.
> If so, I'm a little uncomfortable with persisting a modified copy of
> the normal slot state before that state has actually been published in
> shared memory. It seems to make the state transition somewhat harder
> to follow, since the slot state file no longer necessarily represents
> the current slot state.
>
> Would it be simpler to persist the invalidation separately?
> example, we could write the invalidation cause to a small file such as
> pg_replslot/<slotname>/invalidated and make it durable before
> publishing the invalidation in shared memory. On restart, that file
> would cause the slot to be restored as invalidated with the recorded
> cause. This would keep the normal slot state file as a representation
> of the actual slot state, and would also naturally avoid the race with
> concurrent slot saves.
Your proposal could probably work too. I’m not sure it would be simpler though,
as it would add another on disk state and startup handling. I also could not find
a precedent for introducing such a persistent file in back branches, but I may
have missed one. The proposed patch reuses the existing slot state and format,
which probably makes it more suitable for backpatching.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-27 11:12 Osama Abdul Qader <osamaabdulqader.cs@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Osama Abdul Qader @ 2026-08-27 11:12 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; pgsql-hackers@lists.postgresql.org; amit.kapila16@gmail.com
Thanks for the detailed discussion.
I'm following the reasoning around the durability requirement, particularly
the hot standby case and invalidation causes such as 'inactive_since'.
The separate invalidation file approach makes sense to me as an
alternative, while I also see the advantage of reusing the existing slot
state for backpatching.
I'll hold off on further changes for now and follow the discussion on which
approach is preferred for master implementation.
Regards,
Osama Abdul Qader
On Thu, Aug 27, 2026 at 3:57 PM Bertrand Drouvot <
bertranddrouvot.pg@gmail.com> wrote:
> Hi Horiguchi-san,
>
> On Thu, Aug 27, 2026 at 05:06:47PM +0900, Kyotaro Horiguchi wrote:
> > Hello,
> >
> > At Wed, 26 Aug 2026 13:49:24 +0000, Bertrand Drouvot <
> bertranddrouvot.pg@gmail.com> wrote in
> > > If ReplicationSlotSave() errors before replacing the state file, the
> slot is
> > > invalid in shared memory but still valid on disk. That sounds
> problematic as the
> > > resource horizon computations could stop accounting for the slot,
> remove required
> > > WAL or rows, and then an immediate restart would restore the old valid
> slot image.
> >
> > I've spent some time looking through the related discussions and
> > patches,
>
> Thanks for looking at it!
>
> > For the InvalidatePossiblyObsoleteSlot() case, at least for
> > RS_INVAL_XID_AGE, if the server crashes after the slot is invalidated
> > but before the invalidation is persisted, it seems that the restored
> > slot would still satisfy the same XID-age condition and would
> > eventually be invalidated again by vacuum or checkpoint. Is the main
> > reason for making the invalidation durable here that we don't want to
> > leave the slot valid until that next opportunity?
>
> Yeah, for XID age the condition should still hold after restart. On a
> primary,
> the end of recovery checkpoint should detect it before connections are
> accepted.
>
> On a hot standby, however, connections can be accepted before the next
> successful
> restartpoint, so a restored slot could be used while valid although rows it
> needed may already have been removed.
>
> Also, other causes are not necessarily rediscovered immediately. For
> example,
> inactive_since is reset at startup, so an idle timeout invalidation would
> not
> be detected again until the timeout has elapsed again.
>
> > If so, I'm a little uncomfortable with persisting a modified copy of
> > the normal slot state before that state has actually been published in
> > shared memory. It seems to make the state transition somewhat harder
> > to follow, since the slot state file no longer necessarily represents
> > the current slot state.
> >
> > Would it be simpler to persist the invalidation separately?
>
> > example, we could write the invalidation cause to a small file such as
> > pg_replslot/<slotname>/invalidated and make it durable before
> > publishing the invalidation in shared memory. On restart, that file
> > would cause the slot to be restored as invalidated with the recorded
> > cause. This would keep the normal slot state file as a representation
> > of the actual slot state, and would also naturally avoid the race with
> > concurrent slot saves.
>
> Your proposal could probably work too. I’m not sure it would be simpler
> though,
> as it would add another on disk state and startup handling. I also could
> not find
> a precedent for introducing such a persistent file in back branches, but I
> may
> have missed one. The proposed patch reuses the existing slot state and
> format,
> which probably makes it more suitable for backpatching.
>
> Regards,
>
> --
> Bertrand Drouvot
> PostgreSQL Contributors Team
> RDS Open Source Databases
> Amazon Web Services: https://aws.amazon.com
>
>
>
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-28 10:29 Amit Kapila <amit.kapila16@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2 siblings, 1 reply; 32+ messages in thread
From: Amit Kapila @ 2026-08-28 10:29 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org
On Wed, Aug 26, 2026 at 7:19 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> while reviewing [1], I hit an issue due to the fact that an inactive replication
> slot is marked invalid in shared memory before its new state is persisted.
>
> If ReplicationSlotSave() errors before replacing the state file, the slot is
> invalid in shared memory but still valid on disk. That sounds problematic as the
> resource horizon computations could stop accounting for the slot, remove required
> WAL or rows, and then an immediate restart would restore the old valid slot image.
>
> The same issue exists in synchronize_one_slot(): it copies the invalidation from
> the remote slot into the local synchronized slot before saving it. In that case,
> a save error also prevents a direct retry because the next synchronization sees
> the local slot as already invalid and skips it.
>
Won't the drop_local_obsolete_slots() drop the locally invalidated
slot before trying to synchronize the remote_slot in the next slot?
--
With Regards,
Amit Kapila.
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-08-28 13:32 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: Amit Kapila <amit.kapila16@gmail.com>
0 siblings, 2 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-08-28 13:32 UTC (permalink / raw)
To: Amit Kapila <amit.kapila16@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org
Hi,
On Fri, Aug 28, 2026 at 03:59:32PM +0530, Amit Kapila wrote:
> On Wed, Aug 26, 2026 at 7:19 PM Bertrand Drouvot
> <bertranddrouvot.pg@gmail.com> wrote:
> >
> > while reviewing [1], I hit an issue due to the fact that an inactive replication
> > slot is marked invalid in shared memory before its new state is persisted.
> >
> > If ReplicationSlotSave() errors before replacing the state file, the slot is
> > invalid in shared memory but still valid on disk. That sounds problematic as the
> > resource horizon computations could stop accounting for the slot, remove required
> > WAL or rows, and then an immediate restart would restore the old valid slot image.
> >
> > The same issue exists in synchronize_one_slot(): it copies the invalidation from
> > the remote slot into the local synchronized slot before saving it. In that case,
> > a save error also prevents a direct retry because the next synchronization sees
> > the local slot as already invalid and skips it.
> >
>
> Won't the drop_local_obsolete_slots() drop the locally invalidated
> slot before trying to synchronize the remote_slot in the next slot?
>
It would be if the remote slot were valid and only the local slot invalidated.
Here both are invalidated, so locally_invalidated is false and local_sync_slot_required()
returns true. Thus, drop_local_obsolete_slots() keeps the local slot.
synchronize_one_slot() then sees it already invalidated and skips the save.
The test added by 0002 is meant to cover this. I noticed that v1 restarts
before synchronizing again though, so I changed it in v2 to retry before the
immediate restart and make this case explicit.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v2-0001-Persist-slot-invalidations-before-publishing-them.patch (15.8K, ../../apGN1gY3ATFuxd0C@bdtpg/2-v2-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From 5bb7de2f8080b1f361b612afadb8559e3d325d72 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v2 1/2] Persist slot invalidations before publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 140 +++++++++++++-----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 1 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++
4 files changed, 235 insertions(+), 33 deletions(-)
56.6% src/backend/replication/
41.1% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..b5746eb6283 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotReleaseOnError(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,18 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Release a slot claimed internally for invalidation after an error.
+ */
+static void
+ReplicationSlotReleaseOnError(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1197,29 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+
+ Assert(MyReplicationSlot != NULL);
+ Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR,
+ NameStr(MyReplicationSlot->data.name));
+
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2047,9 +2098,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2108,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2159,8 +2196,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2206,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(
+ invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotReleaseOnError,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2426,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2539,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2519,7 +2565,9 @@ CreateSlotOnDisk(ReplicationSlot *slot)
* Shared functionality between saving and creating a replication slot.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2575,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,9 +2584,11 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
+
LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
/* silence valgrind :( */
@@ -2576,6 +2628,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2669,6 +2735,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..9248a7390a6 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,7 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v2-0002-Persist-synchronized-slot-invalidations-before-pu.patch (7.0K, ../../apGN1gY3ATFuxd0C@bdtpg/3-v2-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 9bb3864fcf1fa8fe1d8b294131aee184058d3738 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v2 2/2] Persist synchronized slot invalidations before
publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. On the next synchronization,
drop_local_obsolete_slots() retains the local slot because the remote slot is
also invalidated. However, synchronize_one_slot() sees the local slot already
invalidated and skips the save, so the failed save is not retried directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, a direct retry,
and an immediate restart that verifies durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 10 +-
src/backend/replication/slot.c | 2 +-
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++++
3 files changed, 129 insertions(+), 8 deletions(-)
10.7% src/backend/replication/logical/
3.2% src/backend/replication/
86.0% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..51c19c60cf9 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,13 +829,9 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
-
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ ReplicationSlotPersistInvalidation(remote_slot->invalidated, false);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index b5746eb6283..02cec90a20b 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1211,7 +1211,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
char path[MAXPGPATH];
Assert(MyReplicationSlot != NULL);
- Assert(MyReplicationSlot->data.persistency == RS_PERSISTENT);
+ Assert(MyReplicationSlot->data.persistency != RS_EPHEMERAL);
Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
index 247d7b00dc1..7f9bfb2b694 100644
--- a/src/test/recovery/t/057_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -122,4 +122,129 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-13 10:31 JoongHyuk Shin <sjh910805@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
1 sibling, 1 reply; 32+ messages in thread
From: JoongHyuk Shin @ 2026-09-13 10:31 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: Amit Kapila <amit.kapila16@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi Bertrand,
I read v2 and have a question about the window the new ordering opens
between two invalidators.
As I understand it, in v2 the first invalidator acquires the slot,
writes the state file with the invalidation, and only after the fsync
sets data.invalidated in shared memory and releases the slot. While the
write is in progress the slot looks valid to everyone else, with the
invalidator's PID as active_pid. If a second invalidator reaches the
same slot in that window (say a restartpoint enforcing
max_slot_wal_keep_size while startup is replaying a wal_level change, or
the other way round), it seems it would take the "slot is in use" path,
so startup might send a recovery conflict to active_pid, or another
process a SIGTERM. What do you think?
--
JoongHyuk Shin
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-17 06:51 Rui Zhao <zhaorui126@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
1 sibling, 0 replies; 32+ messages in thread
From: Rui Zhao @ 2026-09-17 06:51 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: Amit Kapila <amit.kapila16@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
Thanks for v2. I made the save fail with the injection point of 0001,
and the slot comes out of it fine: still valid, released by the
checkpointer, and I can drop it. On master the same failure leaves it
invalid but still owned by the checkpointer, and the next checkpoint
does not fix that:
active|active_pid|invalidation_reason|restart_lsn|wal_status
t|259023|wal_removed||lost
SELECT pg_drop_replication_slot('target_slot');
ERROR: replication slot "target_slot" is active for PID 259023
> If a second invalidator reaches the same slot in that window (say a
> restartpoint enforcing max_slot_wal_keep_size while startup is replaying
> a wal_level change, or the other way round), it seems it would take the
> "slot is in use" path, so startup might send a recovery conflict to
> active_pid, or another process a SIGTERM. What do you think?
Both happen, and the SIGTERM shuts the standby down. 0003 makes the
second invalidator wait for the first one instead, and adds a test with
the two processes in both orders.
The standby has max_slot_wal_keep_size = 1MB, a logical slot 8 segments
of 1MB behind, and the injection point of 0001 attached with 'wait' for
that slot. The primary is restarted with wal_level = replica, so that the
startup process invalidates the slot, and once it waits at the injection
point, CHECKPOINT on the standby:
checkpointer LOG: restartpoint starting: fast wait
checkpointer LOG: terminating process 378422 to release replication
slot "startup_first"
checkpointer DETAIL: The slot's restart_lsn 0/012002D0 exceeds the
limit by 7339312 bytes.
startup LOG: invalidating obsolete replication slot "startup_first"
postmaster LOG: startup process (PID 378422) exited with exit code 1
postmaster LOG: terminating any other active server processes
postmaster LOG: shutting down due to startup process failure
The other way round, the startup process finds the slot held by the
checkpointer and sends it a recovery conflict. The checkpointer errors
out at its next CHECK_FOR_INTERRUPTS(), and the startup process then
invalidates the slot itself:
startup LOG: terminating process 378420 to release replication slot
"checkpointer_first"
checkpointer ERROR: canceling statement due to conflict with recovery
client backend ERROR: checkpoint request failed
startup LOG: invalidating obsolete replication slot "checkpointer_first"
On master the slot is marked invalid under the spinlock at the moment it
is claimed, so the second invalidator sees the invalidation and does
nothing.
0003 takes the slot's io_in_progress_lock before looking at the slot,
and holds it from claiming the slot until the invalidation is published;
0001 holds it for the write only. Whoever comes second then sees either
an invalidated slot or one nobody is working on, and only signals a
process that really uses the slot. If the lock is not free, the function
drops ReplicationSlotControlLock, waits for it, and starts over, the same
way it does around the condition variable sleep. The test fails on v2 as
above and passes with 0003.
The side effects of 0003, as far as I can see:
(a) The second invalidator now waits (wait event ReplicationSlotIO)
instead of signalling; on master it never waited. The wait is one write
and fsync of the slot's state file, and only when two processes go for
the same slot at the same time.
(b) Every pass over the slots now takes and releases each in-use slot's
io_in_progress_lock once. A checkpoint's pass over N valid physical
slots, median of 60 checkpoints:
N v2 0003
10 1.5 us 1.6 us
100 7.2 us 9.0 us
A pass that finds a slot being written waits for that write, without
ReplicationSlotControlLock.
(c) The two locks cannot deadlock: no holder of an io_in_progress_lock
takes ReplicationSlotControlLock, since the callers of SaveSlotToPath()
hold ReplicationSlotAllocationLock at most, and the slot is not in use
yet when CreateSlotOnDisk() writes it. The second invalidator holds
nothing while it waits. Nothing changes for a slot that is really in use:
a walsender's slot, or a temporary slot of an idle session, is signalled
as before.
(d) Not covered: synchronize_one_slot() acquires the slot before it
takes the lock, so a restartpoint that looks at the slot in between still
terminates the slot sync worker. Master has the same window, between
acquiring the slot and publishing; closing it would take a flag in the
slot's shared memory.
I think 0003 is small enough to be backpatched with 0001.
Regards,
Rui
Attachments:
[application/octet-stream] 0003-Wait-for-a-concurrent-invalidation-of-a-slot-instead.patch (14.8K, ../../CAHWVJhF33kdDOE9VVRUM24BD8ALTX5ayuK8roDR8EtpHkmnX7g@mail.gmail.com/2-0003-Wait-for-a-concurrent-invalidation-of-a-slot-instead.patch)
download | inline diff:
From a5c08a9913dea62fcd088be6a403b03afb1de5a5 Mon Sep 17 00:00:00 2001
From: Rui Zhao <zhaorui126@gmail.com>
Date: Mon, 14 Sep 2026 11:47:00 +0800
Subject: [PATCH] Wait for a concurrent invalidation of a slot instead of
terminating it
With the invalidation persisted before it is published, a slot being
invalidated looks like a slot in use for the duration of a file write: it
has an active_pid and no invalidation reason. A second invalidator that
finds it in that state signals its owner as it would signal a walsender.
On a standby the startup process and the checkpointer both invalidate
slots, and the checkpointer signals with SIGTERM, which makes the startup
process exit and takes the standby down.
Examine the slot with its io_in_progress_lock held, and keep that lock
from the moment the slot is claimed until the invalidation is published.
A concurrent invalidator then sees either the invalidation or a slot that
nobody is invalidating. The wait for the lock is done without
ReplicationSlotControlLock, like the other waits in that function.
Add a test with the two processes in both orders.
---
src/backend/replication/logical/slotsync.c | 1 +
src/backend/replication/slot.c | 49 +++-
src/test/recovery/meson.build | 1 +
.../recovery/t/058_slot_invalidation_race.pl | 263 ++++++++++++++++++
4 files changed, 311 insertions(+), 3 deletions(-)
create mode 100644 src/test/recovery/t/058_slot_invalidation_race.pl
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index 51c19c60cf..cd91ab1e9a 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,6 +829,7 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
ReplicationSlotPersistInvalidation(remote_slot->invalidated, false);
ReplicationSlotsComputeRequiredXmin(false);
ReplicationSlotsComputeRequiredLSN();
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 02cec90a20..3805b89560 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1203,6 +1203,8 @@ ReplicationSlotSave(void)
/*
* Persist an invalidated image of the acquired slot before publishing the
* invalidation in shared memory.
+ *
+ * The caller holds the slot's io_in_progress_lock; it is released on return.
*/
void
ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
@@ -1211,6 +1213,8 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
char path[MAXPGPATH];
Assert(MyReplicationSlot != NULL);
+ Assert(LWLockHeldByMeInMode(&MyReplicationSlot->io_in_progress_lock,
+ LW_EXCLUSIVE));
Assert(MyReplicationSlot->data.persistency != RS_EPHEMERAL);
Assert(MyReplicationSlot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
@@ -2016,6 +2020,12 @@ DetermineSlotInvalidationCause(uint32 possible_causes, ReplicationSlot *s,
*
* Acquires the given slot and mark it invalid, if necessary and possible.
*
+ * The slot is examined with its io_in_progress_lock held, and the lock is
+ * kept from the moment the slot is claimed until the invalidation has been
+ * published in shared memory. A concurrent invalidator thus either sees the
+ * invalidation or a slot nobody is invalidating, and does not take the
+ * process persisting the invalidation for a process using the slot.
+ *
* Returns true if the slot was invalidated.
*
* Set *released_lock_out if ReplicationSlotControlLock was released in the
@@ -2056,6 +2066,22 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Getting the io_in_progress_lock may mean waiting for a write of the
+ * slot; don't do that with ReplicationSlotControlLock held.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ LWLockAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE);
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2089,6 +2115,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2134,6 +2161,12 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (active_proc != INVALID_PROC_NUMBER)
{
+ /*
+ * The owner is not invalidating the slot, or we would have seen
+ * the invalidation: it is using the slot.
+ */
+ LWLockRelease(&s->io_in_progress_lock);
+
/*
* Prepare the sleep on the slot's condition variable before
* releasing the lock, to close a possible race condition if the
@@ -2196,8 +2229,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now. Persist its invalidation before
- * publishing it in shared memory.
+ * We hold the slot and its io_in_progress_lock now. Persist the
+ * invalidation before publishing it in shared memory; the lock is
+ * released once that is done.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2589,7 +2623,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ /*
+ * An invalidation is written with the io_in_progress_lock already held by
+ * the caller, from before it claimed the slot; see
+ * InvalidatePossiblyObsoleteSlot(). The lock is released below all the
+ * same.
+ */
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ else
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 9248a7390a..698a934c06 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -66,6 +66,7 @@ tests += {
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_race.pl',
],
},
}
diff --git a/src/test/recovery/t/058_slot_invalidation_race.pl b/src/test/recovery/t/058_slot_invalidation_race.pl
new file mode 100644
index 0000000000..33387885e3
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_race.pl
@@ -0,0 +1,263 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Two processes can try to invalidate the same replication slot at about
+# the same time. On a standby, the startup process invalidates logical
+# slots when the primary disables logical decoding, and the checkpointer
+# invalidates slots whose WAL a restartpoint is about to remove. The one
+# that comes second has to wait for the other, not terminate it.
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('bkp');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'bkp', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+# Set the primary's wal_level and restart it. Going to replica disables
+# logical decoding, which the startup process of the standby applies by
+# invalidating the logical slots of the standby.
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+# Create a logical slot on the standby and make its restart_lsn older than
+# what max_slot_wal_keep_size lets a restartpoint keep. Replay a checkpoint
+# record so that a restartpoint is possible afterwards. The slot's save
+# will stop at the injection point.
+sub create_lagging_slot
+{
+ my ($slot) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot, 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_attach('$injection_point', 'wait',
+ '$slot')});
+}
+
+sub pid_of
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql('postgres',
+ qq{SELECT pid FROM pg_stat_activity
+ WHERE backend_type = '$backend_type'});
+}
+
+# Start a CHECKPOINT on the standby in the background, and return the
+# handle to finish it with.
+sub start_restartpoint
+{
+ my ($out, $err) = @_;
+
+ return IPC::Run::start(
+ [
+ 'psql', '-X', '-d', $standby->connstr('postgres'), '-c',
+ 'CHECKPOINT'
+ ],
+ '>' => $out,
+ '2>' => $err);
+}
+
+# Wait until the process waits for the slot, or is done with it.
+sub wait_for_second_invalidator
+{
+ my ($pid, $done_re, $logstart) = @_;
+ my $primary_lsn = $primary->lsn('write');
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ return
+ if $standby->log_contains($done_re, $logstart);
+ return
+ if $standby->safe_psql(
+ 'postgres',
+ qq{SELECT wait_event IN ('ReplicationSlotIO',
+ 'ReplicationSlotDrop', '$injection_point')
+ OR pg_last_wal_replay_lsn() >= '$primary_lsn'
+ FROM pg_stat_activity WHERE pid = $pid}) eq 't';
+ usleep(100_000);
+ }
+ die "timed out waiting for process $pid to deal with the slot";
+}
+
+# Wake up whoever waits at the injection point, until nobody does and
+# the slot is invalidated.
+sub wake_until_invalidated
+{
+ my ($slot) = @_;
+ my $reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{SELECT count(*) > 0 FROM pg_stat_activity
+ WHERE wait_event = '$injection_point'});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+ $reason = $standby->safe_psql(
+ 'postgres',
+ qq{SELECT invalidation_reason FROM pg_replication_slots
+ WHERE slot_name = '$slot'});
+ last if $reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+ return $reason;
+}
+
+#
+# Checkpointer first, startup process second.
+#
+my $slot = 'checkpointer_first';
+create_lagging_slot($slot);
+my $startup_pid = pid_of('startup');
+my $checkpointer_pid = pid_of('checkpointer');
+my $logstart = -s $standby->logfile;
+
+my ($ckpt_out, $ckpt_err) = ('', '');
+my $ckpt = start_restartpoint(\$ckpt_out, \$ckpt_err);
+$standby->wait_for_event('checkpointer', $injection_point);
+is( $standby->safe_psql(
+ 'postgres',
+ qq{SELECT active_pid = $checkpointer_pid FROM pg_replication_slots
+ WHERE slot_name = '$slot'}),
+ 't',
+ 'checkpointer holds the slot while persisting its invalidation');
+
+set_primary_wal_level('replica');
+wait_for_second_invalidator($startup_pid,
+ qr/invalidating obsolete replication slot "$slot"/, $logstart);
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot "$slot"/,
+ $logstart),
+ 'startup process does not terminate the checkpointer');
+
+my $reason = wake_until_invalidated($slot);
+$ckpt->finish;
+is($ckpt_err, '', 'restartpoint succeeds');
+is($reason, 'wal_removed', 'slot invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $logstart),
+ 'checkpointer sees no recovery conflict');
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+#
+# Startup process first, checkpointer second.
+#
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+$slot = 'startup_first';
+create_lagging_slot($slot);
+$startup_pid = pid_of('startup');
+$checkpointer_pid = pid_of('checkpointer');
+$logstart = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+is( $standby->safe_psql(
+ 'postgres',
+ qq{SELECT active_pid = $startup_pid FROM pg_replication_slots
+ WHERE slot_name = '$slot'}),
+ 't',
+ 'startup process holds the slot while persisting its invalidation');
+
+($ckpt_out, $ckpt_err) = ('', '');
+$ckpt = start_restartpoint(\$ckpt_out, \$ckpt_err);
+$standby->wait_for_log(qr/restartpoint starting/, $logstart);
+wait_for_second_invalidator($checkpointer_pid, qr/restartpoint complete/,
+ $logstart);
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot "$slot"/,
+ $logstart),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+$ckpt->finish;
+is($ckpt_err, '', 'restartpoint succeeds');
+$standby->wait_for_log(qr/invalidating obsolete replication slot "$slot"/,
+ $logstart);
+
+# A startup process that got a SIGTERM exits as soon as it is back in its
+# loop, and takes the standby down with it.
+foreach (1 .. 20)
+{
+ last
+ if $standby->log_contains(
+ qr/startup process \(PID $startup_pid\) exited/, $logstart);
+ usleep(100_000);
+}
+ok( !$standby->log_contains(
+ qr/startup process \(PID $startup_pid\) exited/, $logstart),
+ 'startup process survives');
+ok($standby->is_alive, 'standby is up');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{SELECT invalidation_reason FROM pg_replication_slots
+ WHERE slot_name = '$slot'}),
+ 'wal_level_insufficient',
+ 'slot invalidated by the startup process');
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+ $standby->stop;
+}
+$primary->stop;
+
+done_testing();
--
2.43.7
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-22 08:27 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: JoongHyuk Shin <sjh910805@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-22 08:27 UTC (permalink / raw)
To: JoongHyuk Shin <sjh910805@gmail.com>; +Cc: Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi JoongHyuk,
On Sun, Sep 13, 2026 at 07:31:57PM +0900, JoongHyuk Shin wrote:
> Hi Bertrand,
>
> I read v2 and have a question about the window the new ordering opens
> between two invalidators.
Thanks for looking at it!
> If a second invalidator reaches the
> same slot in that window (say a restartpoint enforcing
> max_slot_wal_keep_size while startup is replaying a wal_level change, or
> the other way round), it seems it would take the "slot is in use" path,
> so startup might send a recovery conflict to active_pid, or another
> process a SIGTERM. What do you think?
You're right, v2 could treat the first invalidator as a regular slot user and
terminate it.
Rui, thanks for the patch! I've incorporated its locking approach and test coverage
in v3, with some adjustments around error cleanup.
Please find v3 attached.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v3-0001-Persist-slot-invalidations-before-publishing-them.patch (29.1K, ../../arI8AEPtq4EtgzWL@bdtpg/2-v3-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From 40e717fb82404bae76edb55f1355cda103df1e09 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v3 1/2] Persist slot invalidations before publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 234 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 587 insertions(+), 49 deletions(-)
45.2% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..8f2b681b626 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,32 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The lock is released on success and left held for error cleanup
+ * otherwise.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2063,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2117,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2138,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2148,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2158,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2186,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2246,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2256,17 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,15 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock. On error, leave the lock held for error cleanup. The lock is
+ * released after successful shared-memory publication.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2628,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2638,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2659,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2686,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2714,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2736,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2753,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2770,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,6 +2797,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v3-0002-Persist-synchronized-slot-invalidations-before-pu.patch (7.6K, ../../arI8AEPtq4EtgzWL@bdtpg/3-v3-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 872516c57a3d37ed794cf3dbc7011e6527f4ee8c Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v3 2/2] Persist synchronized slot invalidations before
publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. On the next synchronization,
drop_local_obsolete_slots() retains the local slot because the remote slot is
also invalidated. However, synchronize_one_slot() sees the local slot already
invalidated and skips the save, so the failed save is not retried directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, a direct retry,
and an immediate restart that verifies durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 18 ++-
src/backend/replication/slot.c | 2 +-
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++++
3 files changed, 137 insertions(+), 8 deletions(-)
17.0% src/backend/replication/logical/
80.5% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..1cfde2cddeb 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,13 +829,10 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
-
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ ReplicationSlotPersistInvalidation(remote_slot->invalidated, false);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
@@ -2007,6 +2004,13 @@ slotsync_failure_callback(int code, Datum arg)
* before it marks itself as finished syncing.
*/
+ /*
+ * A failed invalidation can still hold the slot's I/O lock. Release it
+ * before slot cleanup acquires ReplicationSlotAllocationLock, which
+ * checkpoints hold while acquiring slot I/O locks.
+ */
+ LWLockReleaseAll();
+
/* Make sure active replication slots are released */
if (MyReplicationSlot != NULL)
ReplicationSlotRelease();
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 8f2b681b626..24d7e78e092 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1218,7 +1218,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
ReplicationSlot *slot = MyReplicationSlot;
Assert(slot != NULL);
- Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.persistency != RS_EPHEMERAL);
Assert(slot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
index 247d7b00dc1..7f9bfb2b694 100644
--- a/src/test/recovery/t/057_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -122,4 +122,129 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-22 10:16 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-22 10:16 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Tue, Sep 22, 2026 at 1:58 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi JoongHyuk,
>
> On Sun, Sep 13, 2026 at 07:31:57PM +0900, JoongHyuk Shin wrote:
> > Hi Bertrand,
> >
> > I read v2 and have a question about the window the new ordering opens
> > between two invalidators.
>
> Thanks for looking at it!
>
> > If a second invalidator reaches the
> > same slot in that window (say a restartpoint enforcing
> > max_slot_wal_keep_size while startup is replaying a wal_level change, or
> > the other way round), it seems it would take the "slot is in use" path,
> > so startup might send a recovery conflict to active_pid, or another
> > process a SIGTERM. What do you think?
>
> You're right, v2 could treat the first invalidator as a regular slot user and
> terminate it.
>
> Rui, thanks for the patch! I've incorporated its locking approach and test coverage
> in v3, with some adjustments around error cleanup.
>
> Please find v3 attached.
>
I had a look at 002 to review slotsync path, I had one concern:
+ /*
+ * A failed invalidation can still hold the slot's I/O lock. Release it
+ * before slot cleanup acquires ReplicationSlotAllocationLock, which
+ * checkpoints hold while acquiring slot I/O locks.
+ */
+ LWLockReleaseAll();
+
Could it be problematic to call LWLockReleaseAll() inside a localized
error cleanup callback (PG_ENSURE_ERROR_CLEANUP) rather than waiting
for AbortTransaction or proc_exit? Since the goal is just to avoid
deadlock with the Checkpointer, shouldn't we explicitly release that
one specific lock?
if (MyReplicationSlot != NULL &&
LWLockHeldByMe(&MyReplicationSlot->io_in_progress_lock))
{
LWLockRelease(&MyReplicationSlot->io_in_progress_lock);
}
I don't have an exact scenario to worry about, but it seems like
overkill. Thoughts?
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-23 06:06 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-23 06:06 UTC (permalink / raw)
To: shveta malik <shveta.malik@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Tue, Sep 22, 2026 at 03:46:55PM +0530, shveta malik wrote:
> I had a look at 002 to review slotsync path,
Thanks for looking at it!
> + /*
> + * A failed invalidation can still hold the slot's I/O lock. Release it
> + * before slot cleanup acquires ReplicationSlotAllocationLock, which
> + * checkpoints hold while acquiring slot I/O locks.
> + */
> + LWLockReleaseAll();
> +
>
> Could it be problematic to call LWLockReleaseAll() inside a localized
> error cleanup callback (PG_ENSURE_ERROR_CLEANUP) rather than waiting
> for AbortTransaction or proc_exit? Since the goal is just to avoid
> deadlock with the Checkpointer, shouldn't we explicitly release that
> one specific lock?
> if (MyReplicationSlot != NULL &&
> LWLockHeldByMe(&MyReplicationSlot->io_in_progress_lock))
> {
> LWLockRelease(&MyReplicationSlot->io_in_progress_lock);
> }
>
> I don't have an exact scenario to worry about, but it seems like
> overkill. Thoughts?
Yeah, it's probably better to be specific here.
One concern with the proposed check is that all existing uses of LWLockHeldByMe()
appear to be for assertions or debugging (as documented on top of LWLockHeldByMe()).
Also, releasing an LWLock after ERROR requires restoring the interrupt holdoff
expected by LWLockRelease().
Another possibility would be to make ReplicationSlotPersistInvalidation() always
leave the caller acquired I/O lock held. Slotsync could then release that specific
lock in a PG_CATCH() block, something like:
"
PG_CATCH();
{
HOLD_INTERRUPTS();
LWLockRelease(&slot->io_in_progress_lock);
PG_RE_THROW();
}
PG_END_TRY();
LWLockRelease(&slot->io_in_progress_lock);
"
This would avoid both LWLockReleaseAll() and using LWLockHeldByMe() for normal
control flow. Does that sound preferable?
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-23 06:30 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-23 06:30 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Wed, Sep 23, 2026 at 11:37 AM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi,
>
> On Tue, Sep 22, 2026 at 03:46:55PM +0530, shveta malik wrote:
> > I had a look at 002 to review slotsync path,
>
> Thanks for looking at it!
>
> > + /*
> > + * A failed invalidation can still hold the slot's I/O lock. Release it
> > + * before slot cleanup acquires ReplicationSlotAllocationLock, which
> > + * checkpoints hold while acquiring slot I/O locks.
> > + */
> > + LWLockReleaseAll();
> > +
> >
> > Could it be problematic to call LWLockReleaseAll() inside a localized
> > error cleanup callback (PG_ENSURE_ERROR_CLEANUP) rather than waiting
> > for AbortTransaction or proc_exit? Since the goal is just to avoid
> > deadlock with the Checkpointer, shouldn't we explicitly release that
> > one specific lock?
> > if (MyReplicationSlot != NULL &&
> > LWLockHeldByMe(&MyReplicationSlot->io_in_progress_lock))
> > {
> > LWLockRelease(&MyReplicationSlot->io_in_progress_lock);
> > }
> >
> > I don't have an exact scenario to worry about, but it seems like
> > overkill. Thoughts?
>
> Yeah, it's probably better to be specific here.
>
> One concern with the proposed check is that all existing uses of LWLockHeldByMe()
> appear to be for assertions or debugging (as documented on top of LWLockHeldByMe()).
Oh Okay, I missed this point earlier.
> Also, releasing an LWLock after ERROR requires restoring the interrupt holdoff
> expected by LWLockRelease().
Okay.
> Another possibility would be to make ReplicationSlotPersistInvalidation() always
> leave the caller acquired I/O lock held. Slotsync could then release that specific
> lock in a PG_CATCH() block, something like:
Yes, I agree. I find this approach much better for 2 reasons:
1) The caller has better control over the lock, which makes sense
since it is the one acquiring it.
2) It makes the code much more understandable. Earlier, it took me a
while to figure out exactly where the io_in_progress_lock was getting
released, especially looking at the slotsync patch where it was
acquired right before calling ReplicationSlotPersistInvalidation().
> "
> PG_CATCH();
> {
> HOLD_INTERRUPTS();
> LWLockRelease(&slot->io_in_progress_lock);
> PG_RE_THROW();
> }
> PG_END_TRY();
>
> LWLockRelease(&slot->io_in_progress_lock);
> "
>
> This would avoid both LWLockReleaseAll() and using LWLockHeldByMe() for normal
> control flow. Does that sound preferable?
Yes.
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-23 08:38 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-23 08:38 UTC (permalink / raw)
To: shveta malik <shveta.malik@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Wed, Sep 23, 2026 at 12:00:39PM +0530, shveta malik wrote:
> On Wed, Sep 23, 2026 at 11:37 AM Bertrand Drouvot
> <bertranddrouvot.pg@gmail.com> wrote:
> >
> > Another possibility would be to make ReplicationSlotPersistInvalidation() always
> > leave the caller acquired I/O lock held. Slotsync could then release that specific
> > lock in a PG_CATCH() block, something like:
>
> Yes, I agree. I find this approach much better for 2 reasons:
>
> 1) The caller has better control over the lock, which makes sense
> since it is the one acquiring it.
> 2) It makes the code much more understandable. Earlier, it took me a
> while to figure out exactly where the io_in_progress_lock was getting
> released, especially looking at the slotsync patch where it was
> acquired right before calling ReplicationSlotPersistInvalidation().
>
> > "
> > PG_CATCH();
> > {
> > HOLD_INTERRUPTS();
> > LWLockRelease(&slot->io_in_progress_lock);
> > PG_RE_THROW();
> > }
> > PG_END_TRY();
> >
> > LWLockRelease(&slot->io_in_progress_lock);
> > "
> >
> > This would avoid both LWLockReleaseAll() and using LWLockHeldByMe() for normal
> > control flow. Does that sound preferable?
>
> Yes.
Thanks! Done that way in v4 attached.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v4-0001-Persist-slot-invalidations-before-publishing-them.patch (29.3K, ../../arOP6DEhgKcsNS5b@bdtpg/2-v4-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From 9ddcf170e434f4cc8bdc7be5ab5a6fd329b92ee0 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v4 1/2] Persist slot invalidations before publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 236 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 588 insertions(+), 50 deletions(-)
45.1% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..642babebeab 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,31 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The caller is responsible for releasing the lock.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2062,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2116,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2137,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2147,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2157,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2185,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2245,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2255,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
+ LWLockRelease(&s->io_in_progress_lock);
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2627,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2637,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2658,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2685,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2713,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2735,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2752,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2769,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2796,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v4-0002-Persist-synchronized-slot-invalidations-before-pu.patch (7.6K, ../../arOP6DEhgKcsNS5b@bdtpg/3-v4-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 532ef5eeb81191bf3210e15b39c407401c4b7771 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v4 2/2] Persist synchronized slot invalidations before
publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. On the next synchronization,
drop_local_obsolete_slots() retains the local slot because the remote slot is
also invalidated. However, synchronize_one_slot() sees the local slot already
invalidated and skips the save, so the failed save is not retried directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, a direct retry,
and an immediate restart that verifies durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 28 +++-
src/backend/replication/slot.c | 2 +-
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++++
3 files changed, 148 insertions(+), 7 deletions(-)
20.4% src/backend/replication/logical/
77.2% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..9121a80ec36 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,13 +829,29 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ PG_TRY();
+ {
+ ReplicationSlotPersistInvalidation(remote_slot->invalidated,
+ false);
+ }
+ PG_CATCH();
+ {
+ /*
+ * Release ownership before making the I/O lock available to
+ * concurrent invalidators.
+ */
+ HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
+ ReplicationSlotRelease();
+ LWLockRelease(&slot->io_in_progress_lock);
+ PG_RE_THROW();
+ }
+ PG_END_TRY();
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ LWLockRelease(&slot->io_in_progress_lock);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 642babebeab..65b97dfab91 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1217,7 +1217,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
ReplicationSlot *slot = MyReplicationSlot;
Assert(slot != NULL);
- Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.persistency != RS_EPHEMERAL);
Assert(slot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
index 247d7b00dc1..7f9bfb2b694 100644
--- a/src/test/recovery/t/057_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -122,4 +122,129 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-23 10:37 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-23 10:37 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Wed, Sep 23, 2026 at 2:08 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi,
>
> On Wed, Sep 23, 2026 at 12:00:39PM +0530, shveta malik wrote:
> > On Wed, Sep 23, 2026 at 11:37 AM Bertrand Drouvot
> > <bertranddrouvot.pg@gmail.com> wrote:
> > >
> > > Another possibility would be to make ReplicationSlotPersistInvalidation() always
> > > leave the caller acquired I/O lock held. Slotsync could then release that specific
> > > lock in a PG_CATCH() block, something like:
> >
> > Yes, I agree. I find this approach much better for 2 reasons:
> >
> > 1) The caller has better control over the lock, which makes sense
> > since it is the one acquiring it.
> > 2) It makes the code much more understandable. Earlier, it took me a
> > while to figure out exactly where the io_in_progress_lock was getting
> > released, especially looking at the slotsync patch where it was
> > acquired right before calling ReplicationSlotPersistInvalidation().
> >
> > > "
> > > PG_CATCH();
> > > {
> > > HOLD_INTERRUPTS();
> > > LWLockRelease(&slot->io_in_progress_lock);
> > > PG_RE_THROW();
> > > }
> > > PG_END_TRY();
> > >
> > > LWLockRelease(&slot->io_in_progress_lock);
> > > "
> > >
> > > This would avoid both LWLockReleaseAll() and using LWLockHeldByMe() for normal
> > > control flow. Does that sound preferable?
> >
> > Yes.
>
> Thanks! Done that way in v4 attached.
>
Can we create the helper function for this logic as it is not core to
the sync-function. Please use the attached patch if you agree, and
feel free to change comments as you see apt.
This approach is better for one more reason: v3 was only releasing the
I/O lock locally during the API call (via slotsync_failure_callback),
leaving the slot-sync worker to rely on top-level proc_exit perhaps.
Current approach ensures consistent lock lifecycle handling for both
the worker and the API.
thanks
Shhveta
Attachments:
[application/octet-stream] 0001-helper-function.patch (2.8K, ../../CAJpy0uAiqT-__cV-MWNCQTXpAjcVDNh53+L2-DzZnYH8O5K1Vg@mail.gmail.com/2-0001-helper-function.patch)
download | inline diff:
From 557ecbbead1ed01c0d11763fe65f02b7db6184c1 Mon Sep 17 00:00:00 2001
From: Shveta Malik <shveta.malik@gmail.com>
Date: Wed, 23 Sep 2026 15:46:51 +0530
Subject: [PATCH] helper function
---
src/backend/replication/logical/slotsync.c | 63 ++++++++++++++--------
1 file changed, 42 insertions(+), 21 deletions(-)
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index 9121a80ec36..b4731687865 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -671,6 +671,47 @@ reserve_wal_for_local_slot(XLogRecPtr restart_lsn)
LWLockRelease(ReplicationSlotAllocationLock);
}
+/*
+ * Persist the invalidated state of a synchronized slot to disk.
+ *
+ * This encapsulates the required I/O lock management and error handling.
+ * If the disk write fails, we must explicitly release the I/O lock
+ * before re-throwing the error to avoid deadlock with concurrent
+ * Checkpointer processes waiting for the I/O lock while holding
+ * ReplicationSlotAllocationLock.
+ */
+static void
+persist_slot_invalidation(ReplicationSlot *slot, ReplicationSlotInvalidationCause cause)
+{
+ Assert(slot != NULL);
+
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ PG_TRY();
+ {
+ /*
+ * It persists the invalidated state to disk before publishing it in
+ * shared memory to ensure state durability on crash.
+ */
+ ReplicationSlotPersistInvalidation(cause,
+ false);
+ }
+ PG_CATCH();
+ {
+ /*
+ * Release ownership before making the I/O lock available to
+ * concurrent invalidators.
+ */
+ HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
+ ReplicationSlotRelease();
+ LWLockRelease(&slot->io_in_progress_lock);
+ PG_RE_THROW();
+ }
+ PG_END_TRY();
+
+ LWLockRelease(&slot->io_in_progress_lock);
+}
+
/*
* If the remote restart_lsn and catalog_xmin have caught up with the
* local ones, then update the LSNs and persist the local synced slot for
@@ -829,27 +870,7 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
-
- PG_TRY();
- {
- ReplicationSlotPersistInvalidation(remote_slot->invalidated,
- false);
- }
- PG_CATCH();
- {
- /*
- * Release ownership before making the I/O lock available to
- * concurrent invalidators.
- */
- HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
- ReplicationSlotRelease();
- LWLockRelease(&slot->io_in_progress_lock);
- PG_RE_THROW();
- }
- PG_END_TRY();
-
- LWLockRelease(&slot->io_in_progress_lock);
+ persist_slot_invalidation(slot, remote_slot->invalidated);
ReplicationSlotsComputeRequiredXmin(false);
ReplicationSlotsComputeRequiredLSN();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-23 16:05 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-23 16:05 UTC (permalink / raw)
To: shveta malik <shveta.malik@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Wed, Sep 23, 2026 at 04:07:02PM +0530, shveta malik wrote:
> Can we create the helper function for this logic as it is not core to
> the sync-function. Please use the attached patch if you agree, and
> feel free to change comments as you see apt.
I was initially a bit skeptical about adding a helper with only one caller, as it
seemed to just move the PG_TRY() block elsewhere without reducing duplication.
On second thought, I agree it makes sense here, as it keeps the lock acquisition
and required ERROR cleanup ordering together. There are also similar "one caller"
helpers wrapping PG_TRY() blocks, such as start_table_sync() and start_sequence_sync().
+static void
+persist_slot_invalidation(ReplicationSlot *slot, ReplicationSlotInvalidationCause cause)
+{
I changed it slightly to derive slot from MyReplicationSlot, ensuring that the I/O
lock belongs to the same slot that ReplicationSlotPersistInvalidation() and
and ReplicationSlotRelease() are using. This is also consistent with other operations
on the currently acquired slot.
Please find v5 attached.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v5-0001-Persist-slot-invalidations-before-publishing-them.patch (29.3K, ../../arP4wN5HW8h0Sr90@bdtpg/2-v5-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From 178afb70032180b30aa50452d915d5fccea991c5 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v5 1/2] Persist slot invalidations before publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A write failure now leaves both the shared slot and its disk image valid.
Ensure that errors also release ownership claimed for inactive slots while
preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add an injection-point test covering save failure, subsequent checkpointing,
and immediate restart. This changes neither the on disk slot format nor the
ReplicationSlot shared memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 236 +++++++++++----
src/include/replication/slot.h | 2 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 588 insertions(+), 50 deletions(-)
45.1% src/backend/replication/
53.5% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..642babebeab 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,17 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
+static void ReplicationSlotInvalidationErrorCleanup(int code, Datum arg);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +773,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +789,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +822,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +837,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -850,6 +867,22 @@ ReplicationSlotRelease(void)
}
}
+/*
+ * Roll back an internal invalidation after an error.
+ *
+ * On ERROR, release slot ownership before normal error cleanup calls
+ * LWLockReleaseAll() to release the I/O lock. During process exit,
+ * shmem_exit() has already released LWLocks before invoking this callback.
+ */
+static void
+ReplicationSlotInvalidationErrorCleanup(int code, Datum arg)
+{
+ ReplicationSlot *slot = (ReplicationSlot *) DatumGetPointer(arg);
+
+ if (MyReplicationSlot == slot)
+ ReplicationSlotReleaseInternal(false);
+}
+
/*
* Cleanup temporary slots created in current session.
*
@@ -1168,7 +1201,31 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory. The caller must own the slot and hold its
+ * I/O lock. The caller is responsible for releasing the lock.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
}
/*
@@ -2005,6 +2062,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2116,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2137,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2147,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2157,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2185,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2245,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,9 +2255,18 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ PG_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+ {
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED);
+ }
+ PG_END_ENSURE_ERROR_CLEANUP(ReplicationSlotInvalidationErrorCleanup,
+ PointerGetDatum(s));
+
+ /* Let caller know */
+ invalidated = true;
+ LWLockRelease(&s->io_in_progress_lock);
ReplicationSlotRelease();
ReportSlotInvalidation(invalidation_cause, false, active_pid,
@@ -2380,7 +2475,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2588,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2612,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2627,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2637,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2658,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2685,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2713,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2735,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2752,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2769,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2796,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..80d48020a87 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,8 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v5-0002-Persist-synchronized-slot-invalidations-before-pu.patch (8.2K, ../../arP4wN5HW8h0Sr90@bdtpg/3-v5-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 29748bb0f86e40ad2b07eaac9b300d47739dce40 Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v5 2/2] Persist synchronized slot invalidations before
publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. On the next synchronization,
drop_local_obsolete_slots() retains the local slot because the remote slot is
also invalidated. However, synchronize_one_slot() sees the local slot already
invalidated and skips the save, so the failed save is not retried directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, a direct retry,
and an immediate restart that verifies durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 42 +++++-
src/backend/replication/slot.c | 2 +-
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++++++++++++
3 files changed, 161 insertions(+), 8 deletions(-)
26.2% src/backend/replication/logical/
71.6% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..f45e25f38cb 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -671,6 +671,38 @@ reserve_wal_for_local_slot(XLogRecPtr restart_lsn)
LWLockRelease(ReplicationSlotAllocationLock);
}
+/*
+ * Persist the invalidated state of the acquired synchronized slot.
+ *
+ * On ERROR, release slot ownership before making the I/O lock available to
+ * concurrent invalidators. The I/O lock must be released before higher-level
+ * error cleanup, which may acquire ReplicationSlotAllocationLock.
+ */
+static void
+persist_slot_invalidation(ReplicationSlotInvalidationCause cause)
+{
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ PG_TRY();
+ {
+ ReplicationSlotPersistInvalidation(cause, false);
+ }
+ PG_CATCH();
+ {
+ HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
+ ReplicationSlotRelease();
+ LWLockRelease(&slot->io_in_progress_lock);
+ PG_RE_THROW();
+ }
+ PG_END_TRY();
+
+ LWLockRelease(&slot->io_in_progress_lock);
+}
+
/*
* If the remote restart_lsn and catalog_xmin have caught up with the
* local ones, then update the LSNs and persist the local synced slot for
@@ -829,13 +861,9 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
-
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ persist_slot_invalidation(remote_slot->invalidated);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 642babebeab..65b97dfab91 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1217,7 +1217,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
ReplicationSlot *slot = MyReplicationSlot;
Assert(slot != NULL);
- Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.persistency != RS_EPHEMERAL);
Assert(slot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
index 247d7b00dc1..7f9bfb2b694 100644
--- a/src/test/recovery/t/057_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -122,4 +122,129 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-24 03:49 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-24 03:49 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Wed, Sep 23, 2026 at 9:35 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi,
>
> On Wed, Sep 23, 2026 at 04:07:02PM +0530, shveta malik wrote:
> > Can we create the helper function for this logic as it is not core to
> > the sync-function. Please use the attached patch if you agree, and
> > feel free to change comments as you see apt.
>
> I was initially a bit skeptical about adding a helper with only one caller, as it
> seemed to just move the PG_TRY() block elsewhere without reducing duplication.
>
> On second thought, I agree it makes sense here, as it keeps the lock acquisition
> and required ERROR cleanup ordering together. There are also similar "one caller"
> helpers wrapping PG_TRY() blocks, such as start_table_sync() and start_sequence_sync().
>
> +static void
> +persist_slot_invalidation(ReplicationSlot *slot, ReplicationSlotInvalidationCause cause)
> +{
>
> I changed it slightly to derive slot from MyReplicationSlot, ensuring that the I/O
> lock belongs to the same slot that ReplicationSlotPersistInvalidation() and
> and ReplicationSlotRelease() are using. This is also consistent with other operations
> on the currently acquired slot.
Yes, that makes sense.
> Please find v5 attached.
Code changes looks good to me. Regarding the test, I have one trivial
comment: I think we should cover all the cases:
a) Failed sync: Memory is NOT updated.
b) Successful sync: Memory IS updated.
c) Server restart: Memory is STILL updated (because it was recovered
from disk).
I think b) is not covered. It will be good to check the memory state
before standby-stop here:
$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
$standby->stop('immediate');
$standby->start;
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-24 05:25 shveta malik <shveta.malik@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-24 05:25 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
I had a look at patch001 as well. I have 2 questions:
1)
Would it be better to use a PG_TRY/PG_CATCH block in patch 001,
similar to patch 002? Currently, the slot is released via
PG_ENSURE_ERROR_CLEANUP, while we rely on top-level error cleanup for
lock-release. Using TRY/CATCH would let us explicitly release both the
slot and I/O lock together, making the error handling consistent
across both patches. We could even reuse persist_slot_invalidation()
with a small change to pass update_inactive_since from the caller.
2)
+ /* Let caller know */
+ invalidated = true;
+ LWLockRelease(&s->io_in_progress_lock);
ReplicationSlotRelease();
Wouldn't it be better (and safer) to release the slot before releasing
the I/O lock?
Currently, concurrent invalidators are protected by the
'invalidation_cause == RS_INVAL_NONE' check after acquiring the lock.
But releasing the slot first would close this race window entirely. It
would also make the order consistent with Patch 002 and the
error-handling flow in Patch 001 itself.
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-24 09:41 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 2 replies; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-24 09:41 UTC (permalink / raw)
To: shveta malik <shveta.malik@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Thu, Sep 24, 2026 at 10:55:03AM +0530, shveta malik wrote:
> I had a look at patch001 as well. I have 2 questions:
>
> 1)
> Would it be better to use a PG_TRY/PG_CATCH block in patch 001,
> similar to patch 002? Currently, the slot is released via
> PG_ENSURE_ERROR_CLEANUP, while we rely on top-level error cleanup for
> lock-release. Using TRY/CATCH would let us explicitly release both the
> slot and I/O lock together, making the error handling consistent
> across both patches. We could even reuse persist_slot_invalidation()
> with a small change to pass update_inactive_since from the caller.
Yeah, makes sense. I changed 0001 to use PG_TRY/PG_CATCH and moved the common
cleanup into ReplicationSlotPersistInvalidation(), with update_inactive_since
passed by the caller.
The I/O lock is still acquired by each caller because 0001 must hold it before
claiming the inactive slot to serialize concurrent internal invalidators.
> 2)
> + /* Let caller know */
> + invalidated = true;
> + LWLockRelease(&s->io_in_progress_lock);
> ReplicationSlotRelease();
>
> Wouldn't it be better (and safer) to release the slot before releasing
> the I/O lock?
>
> Currently, concurrent invalidators are protected by the
> 'invalidation_cause == RS_INVAL_NONE' check after acquiring the lock.
> But releasing the slot first would close this race window entirely. It
> would also make the order consistent with Patch 002 and the
> error-handling flow in Patch 001 itself.
The current ordering should be safe because the invalidation has already been
published, so a concurrent invalidator exits before considering active_proc.
That said, I agree that releasing the slot first means this ordering no longer
relies on that check and makes the success and error paths consistent. So, done
in the attached.
It also adds the check you suggested for the synchronized slot's shared memory
state after a successful synchronization.
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
Attachments:
[text/x-diff] v6-0001-Persist-slot-invalidations-before-publishing-them.patch (28.9K, ../../arTwLinbxvVkEqu2@bdtpg/2-v6-0001-Persist-slot-invalidations-before-publishing-them.patch)
download | inline diff:
From 22308850ac3dbdbd3724286127bd017f73cb369d Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:34:51 +0000
Subject: [PATCH v6 1/2] Persist slot invalidations before publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
InvalidatePossiblyObsoleteSlot() marks an inactive replication slot invalid
in shared memory before saving it. If the save fails, or the server crashes
before it completes, startup can restore a valid slot after resources required
by that slot have been removed.
Add ReplicationSlotPersistInvalidation(), which writes and fsyncs an invalidated
copy while the shared slot remains valid. Hold io_in_progress_lock until the
invalidation is published so checkpoints and other slot savers cannot persist
a stale image after publication.
A failed save now leaves both the shared slot and its disk image valid. On
ERROR, release ownership claimed for an inactive slot before releasing the I/O
lock while preserving inactive_since.
Also serialize concurrent internal invalidators with the slot's I/O lock. This
makes a second invalidator wait instead of treating the first as a regular slot
user and terminating it.
Add tests using an injection point to cover failed saves, subsequent
checkpointing and immediate restart, and concurrent internal invalidators.
This changes neither the on disk slot format nor the ReplicationSlot shared
memory layout.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 14
---
src/backend/replication/slot.c | 229 +++++++++++----
src/include/replication/slot.h | 3 +
src/test/recovery/meson.build | 2 +
.../t/057_replslot_invalidation_durability.pl | 125 ++++++++
.../t/058_slot_invalidation_concurrency.pl | 273 ++++++++++++++++++
5 files changed, 582 insertions(+), 50 deletions(-)
43.6% src/backend/replication/
54.9% src/test/recovery/t/
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index 63ce6d27885..ef54833e0f9 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -185,13 +185,16 @@ static SyncStandbySlotsConfigData *synchronized_standby_slots_config;
static XLogRecPtr ss_oldest_flush_lsn = InvalidXLogRecPtr;
static void ReplicationSlotShmemExit(int code, Datum arg);
+static void ReplicationSlotReleaseInternal(bool update_inactive_since);
static bool IsSlotForConflictCheck(const char *name);
static void ReplicationSlotDropPtr(ReplicationSlot *slot);
/* internal persistency functions */
static void RestoreSlotFromDisk(const char *name);
static void CreateSlotOnDisk(ReplicationSlot *slot);
-static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel);
+static void SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn);
/*
* Register shared memory space needed for replication slots.
@@ -769,6 +772,15 @@ retry:
*/
void
ReplicationSlotRelease(void)
+{
+ ReplicationSlotReleaseInternal(true);
+}
+
+/*
+ * Release the replication slot, optionally preserving inactive_since.
+ */
+static void
+ReplicationSlotReleaseInternal(bool update_inactive_since)
{
ReplicationSlot *slot = MyReplicationSlot;
char *slotname = NULL; /* keep compiler quiet */
@@ -776,6 +788,7 @@ ReplicationSlotRelease(void)
TimestampTz now = 0;
Assert(slot != NULL && slot->active_proc != INVALID_PROC_NUMBER);
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
is_logical = SlotIsLogical(slot);
@@ -808,10 +821,12 @@ ReplicationSlotRelease(void)
}
/*
- * Set the time since the slot has become inactive. We get the current
- * time beforehand to avoid system call while holding the spinlock.
+ * Set the time since the slot has become inactive, unless the caller
+ * needs to preserve it. Get the current time beforehand to avoid a
+ * system call while holding the spinlock.
*/
- now = GetCurrentTimestamp();
+ if (update_inactive_since)
+ now = GetCurrentTimestamp();
if (slot->data.persistency == RS_PERSISTENT)
{
@@ -821,11 +836,12 @@ ReplicationSlotRelease(void)
*/
SpinLockAcquire(&slot->mutex);
slot->active_proc = INVALID_PROC_NUMBER;
- ReplicationSlotSetInactiveSince(slot, now, false);
+ if (update_inactive_since)
+ ReplicationSlotSetInactiveSince(slot, now, false);
SpinLockRelease(&slot->mutex);
ConditionVariableBroadcast(&slot->active_cv);
}
- else
+ else if (update_inactive_since)
ReplicationSlotSetInactiveSince(slot, now, true);
MyReplicationSlot = NULL;
@@ -1168,7 +1184,46 @@ ReplicationSlotSave(void)
Assert(MyReplicationSlot != NULL);
sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(MyReplicationSlot->data.name));
- SaveSlotToPath(MyReplicationSlot, path, ERROR);
+ SaveSlotToPath(MyReplicationSlot, path, ERROR, RS_INVAL_NONE, false);
+}
+
+/*
+ * Persist an invalidated image of the acquired slot before publishing the
+ * invalidation in shared memory.
+ *
+ * The caller must own the slot and hold its I/O lock. On success, the caller
+ * retains both. On ERROR, release slot ownership before the I/O lock and
+ * update inactive_since only if requested.
+ */
+void
+ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn,
+ bool update_inactive_since)
+{
+ char path[MAXPGPATH];
+ ReplicationSlot *slot = MyReplicationSlot;
+
+ Assert(slot != NULL);
+ Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+ Assert(cause != RS_INVAL_NONE);
+ Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock, LW_EXCLUSIVE));
+
+ sprintf(path, "%s/%s", PG_REPLSLOT_DIR, NameStr(slot->data.name));
+
+ PG_TRY();
+ {
+ SaveSlotToPath(slot, path, ERROR, cause, clear_restart_lsn);
+ }
+ PG_CATCH();
+ {
+ HOLD_INTERRUPTS(); /* match the upcoming RESUME_INTERRUPTS */
+ ReplicationSlotReleaseInternal(update_inactive_since);
+ LWLockRelease(&slot->io_in_progress_lock);
+ PG_RE_THROW();
+ }
+ PG_END_TRY();
}
/*
@@ -2005,6 +2060,49 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
break;
}
+ /*
+ * Serializing on the slot's I/O lock ensures that an internal
+ * invalidator cannot be mistaken for a process using the slot. Avoid
+ * waiting for the lock while holding ReplicationSlotControlLock.
+ */
+ if (!LWLockConditionalAcquire(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ {
+ /*
+ * Avoid waiting for an unrelated slot save. The check after
+ * acquiring the lock remains authoritative.
+ */
+ if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
+ now = GetCurrentTimestamp();
+
+ SpinLockAcquire(&s->mutex);
+
+ if (s->data.invalidated == RS_INVAL_NONE)
+ invalidation_cause = DetermineSlotInvalidationCause(possible_causes,
+ s, oldestLSN,
+ dboid,
+ snapshotConflictHorizon,
+ &inactive_since, now);
+
+ SpinLockRelease(&s->mutex);
+
+ if (invalidation_cause == RS_INVAL_NONE)
+ {
+ if (released_lock)
+ LWLockRelease(ReplicationSlotControlLock);
+
+ break;
+ }
+
+ LWLockRelease(ReplicationSlotControlLock);
+ released_lock = true;
+
+ if (LWLockAcquireOrWait(&s->io_in_progress_lock, LW_EXCLUSIVE))
+ LWLockRelease(&s->io_in_progress_lock);
+
+ LWLockAcquire(ReplicationSlotControlLock, LW_SHARED);
+ continue;
+ }
+
if (possible_causes & RS_INVAL_IDLE_TIMEOUT)
{
/*
@@ -2016,10 +2114,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
/*
* Check if the slot needs to be invalidated. If it needs to be
- * invalidated, and is not currently acquired, acquire it and mark it
- * as having been invalidated. We do this with the spinlock held to
- * avoid race conditions -- for example the restart_lsn could move
- * forward, or the slot could be dropped.
+ * invalidated and is not currently acquired, acquire it. We do this
+ * with the spinlock held to avoid races where restart_lsn moves
+ * forward or the slot is dropped.
*/
SpinLockAcquire(&s->mutex);
@@ -2038,6 +2135,7 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
if (invalidation_cause == RS_INVAL_NONE)
{
SpinLockRelease(&s->mutex);
+ LWLockRelease(&s->io_in_progress_lock);
if (released_lock)
LWLockRelease(ReplicationSlotControlLock);
break;
@@ -2047,9 +2145,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
active_proc = s->active_proc;
/*
- * If the slot can be acquired, do so and mark it invalidated
- * immediately. Otherwise we'll signal the owning process, below, and
- * retry.
+ * If the slot can be acquired, do so. Otherwise we'll signal the
+ * owning process, below, and retry.
*
* Note: Unlike other slot attributes, slot's inactive_since can't be
* changed until the acquired slot is released or the owning process
@@ -2058,22 +2155,9 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
*/
if (active_proc == INVALID_PROC_NUMBER)
{
+ Assert(s->data.persistency == RS_PERSISTENT);
MyReplicationSlot = s;
s->active_proc = MyProcNumber;
- s->data.invalidated = invalidation_cause;
-
- /*
- * XXX: We should consider not overwriting restart_lsn and instead
- * just rely on .invalidated.
- */
- if (invalidation_cause == RS_INVAL_WAL_REMOVED)
- {
- s->data.restart_lsn = InvalidXLogRecPtr;
- s->last_saved_restart_lsn = InvalidXLogRecPtr;
- }
-
- /* Let caller know */
- invalidated = true;
}
else
{
@@ -2099,11 +2183,11 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
{
/*
* Prepare the sleep on the slot's condition variable before
- * releasing the lock, to close a possible race condition if the
- * slot is released before the sleep below.
+ * releasing either lock.
*/
ConditionVariablePrepareToSleep(&s->active_cv);
+ LWLockRelease(&s->io_in_progress_lock);
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
@@ -2159,8 +2243,8 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
else
{
/*
- * We hold the slot now and have already invalidated it; flush it
- * to ensure that state persists.
+ * We hold the slot now. Persist its invalidation before
+ * publishing it in shared memory.
*
* Don't want to hold ReplicationSlotControlLock across file
* system operations, so release it now but be sure to tell caller
@@ -2169,10 +2253,14 @@ InvalidatePossiblyObsoleteSlot(uint32 possible_causes,
LWLockRelease(ReplicationSlotControlLock);
released_lock = true;
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ ReplicationSlotPersistInvalidation(invalidation_cause,
+ invalidation_cause == RS_INVAL_WAL_REMOVED,
+ false);
+
+ /* Let caller know */
+ invalidated = true;
ReplicationSlotRelease();
+ LWLockRelease(&s->io_in_progress_lock);
ReportSlotInvalidation(invalidation_cause, false, active_pid,
slotname, restart_lsn,
@@ -2380,7 +2468,7 @@ CheckPointReplicationSlots(bool is_shutdown)
if (s->last_saved_restart_lsn != s->data.restart_lsn)
last_saved_restart_lsn_updated = true;
- SaveSlotToPath(s, path, LOG);
+ SaveSlotToPath(s, path, LOG, RS_INVAL_NONE, false);
}
LWLockRelease(ReplicationSlotAllocationLock);
@@ -2493,7 +2581,7 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/* Write the actual state file. */
slot->dirty = true; /* signal that we really need to write */
- SaveSlotToPath(slot, tmppath, ERROR);
+ SaveSlotToPath(slot, tmppath, ERROR, RS_INVAL_NONE, false);
/* Rename the directory into place. */
if (rename(tmppath, path) != 0)
@@ -2517,9 +2605,14 @@ CreateSlotOnDisk(ReplicationSlot *slot)
/*
* Shared functionality between saving and creating a replication slot.
+ *
+ * When invalidation_cause is set, the caller has already acquired the slot's
+ * I/O lock and remains responsible for releasing it.
*/
static void
-SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
+SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel,
+ ReplicationSlotInvalidationCause invalidation_cause,
+ bool clear_restart_lsn)
{
char tmppath[MAXPGPATH];
char path[MAXPGPATH];
@@ -2527,6 +2620,9 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
ReplicationSlotOnDisk cp;
bool was_dirty;
+ Assert(!clear_restart_lsn || invalidation_cause == RS_INVAL_WAL_REMOVED);
+ Assert(invalidation_cause == RS_INVAL_NONE || elevel >= ERROR);
+
/* first check whether there's something to write out */
SpinLockAcquire(&slot->mutex);
was_dirty = slot->dirty;
@@ -2534,10 +2630,16 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
/* and don't do anything if there's nothing to write */
- if (!was_dirty)
+ if (!was_dirty && invalidation_cause == RS_INVAL_NONE)
return;
- LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ if (invalidation_cause != RS_INVAL_NONE)
+ Assert(LWLockHeldByMeInMode(&slot->io_in_progress_lock,
+ LW_EXCLUSIVE));
+ else
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+
+ INJECTION_POINT("replication-slot-save-error", NameStr(slot->data.name));
/* silence valgrind :( */
memset(&cp, 0, sizeof(ReplicationSlotOnDisk));
@@ -2549,14 +2651,14 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
if (fd < 0)
{
/*
- * If not an ERROR, then release the lock before returning. In case
- * of an ERROR, the error recovery path automatically releases the
- * lock, but no harm in explicitly releasing even in that case. Note
- * that LWLockRelease() could affect errno.
+ * Keep a caller-owned lock until its error cleanup has rolled back
+ * any associated shared-memory state. Note that LWLockRelease() could
+ * affect errno.
*/
int save_errno = errno;
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
(errcode_for_file_access(),
@@ -2576,6 +2678,20 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
SpinLockRelease(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(cp.slotdata.invalidated == RS_INVAL_NONE);
+
+ cp.slotdata.invalidated = invalidation_cause;
+
+ /*
+ * XXX: We should consider not overwriting restart_lsn and instead
+ * just rely on .invalidated.
+ */
+ if (clear_restart_lsn)
+ cp.slotdata.restart_lsn = InvalidXLogRecPtr;
+ }
+
COMP_CRC32C(cp.checksum,
(char *) (&cp) + ReplicationSlotOnDiskNotChecksummedSize,
ReplicationSlotOnDiskChecksummedSize);
@@ -2590,7 +2706,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
/* if write didn't set errno, assume problem is no disk space */
errno = save_errno ? save_errno : ENOSPC;
@@ -2611,7 +2728,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
pgstat_report_wait_end();
CloseTransientFile(fd);
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2627,7 +2745,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2643,7 +2762,8 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
int save_errno = errno;
unlink(tmppath);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
errno = save_errno;
ereport(elevel,
@@ -2669,13 +2789,22 @@ SaveSlotToPath(ReplicationSlot *slot, const char *dir, int elevel)
* already and remember the confirmed_flush LSN value.
*/
SpinLockAcquire(&slot->mutex);
+ if (invalidation_cause != RS_INVAL_NONE)
+ {
+ Assert(slot->data.invalidated == RS_INVAL_NONE);
+
+ slot->data.invalidated = invalidation_cause;
+ if (clear_restart_lsn)
+ slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
if (!slot->just_dirtied)
slot->dirty = false;
slot->last_saved_confirmed_flush = cp.slotdata.confirmed_flush;
slot->last_saved_restart_lsn = cp.slotdata.restart_lsn;
SpinLockRelease(&slot->mutex);
- LWLockRelease(&slot->io_in_progress_lock);
+ if (invalidation_cause == RS_INVAL_NONE)
+ LWLockRelease(&slot->io_in_progress_lock);
}
/*
diff --git a/src/include/replication/slot.h b/src/include/replication/slot.h
index 9b29444cbca..885d3af236c 100644
--- a/src/include/replication/slot.h
+++ b/src/include/replication/slot.h
@@ -344,6 +344,9 @@ extern void ReplicationSlotAcquire(const char *name, bool nowait,
extern void ReplicationSlotRelease(void);
extern void ReplicationSlotCleanup(bool synced_only);
extern void ReplicationSlotSave(void);
+extern void ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
+ bool clear_restart_lsn,
+ bool update_inactive_since);
extern void ReplicationSlotMarkDirty(void);
/* misc stuff */
diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build
index 72113c5ac6e..e74b547a961 100644
--- a/src/test/recovery/meson.build
+++ b/src/test/recovery/meson.build
@@ -65,6 +65,8 @@ tests += {
't/054_unlogged_sequence_promotion.pl',
't/055_cascade_reconnect.pl',
't/056_standby_snapshot_export.pl',
+ 't/057_replslot_invalidation_durability.pl',
+ 't/058_slot_invalidation_concurrency.pl',
],
},
}
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
new file mode 100644
index 00000000000..247d7b00dc1
--- /dev/null
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -0,0 +1,125 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test that replication slot invalidation is persisted before it is published.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $node = PostgreSQL::Test::Cluster->new('primary');
+$node->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$node->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+min_wal_size = 2MB
+max_wal_size = 64MB
+wal_keep_size = 0
+max_slot_wal_keep_size = -1
+log_checkpoints = on
+));
+$node->start;
+
+if (!$node->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$node->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$node->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('target_slot', true)});
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+my ($restart_lsn, $restart_segment) = split(
+ /\|/,
+ $node->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn, pg_walfile_name(restart_lsn)
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}));
+my $restart_segment_path = $node->data_dir . "/pg_wal/$restart_segment";
+my $inactive_since = $node->safe_psql(
+ 'postgres',
+ q{
+SELECT inactive_since
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+});
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$node->reload;
+$node->advance_wal(8);
+
+my $current_segment = $node->safe_psql('postgres',
+ 'SELECT pg_walfile_name(pg_current_wal_lsn())');
+isnt($current_segment, $restart_segment,
+ 'target slot requires an older WAL segment');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists before invalidation");
+
+$node->safe_psql(
+ 'postgres', q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'target_slot')
+});
+
+my ($ret, $stdout, $stderr) = $node->psql('postgres', 'CHECKPOINT');
+like(
+ $stderr,
+ qr/checkpoint request failed/,
+ 'injected slot save error failed the checkpoint');
+
+$node->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn',
+ inactive_since = '$inactive_since'::timestamptz
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t|t',
+ 'failed save leaves the valid slot unchanged');
+ok( -f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the failed checkpoint"
+);
+
+$node->append_conf('postgresql.conf', 'max_slot_wal_keep_size = -1');
+$node->reload;
+$node->safe_psql('postgres', 'CHECKPOINT');
+
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment survives the next checkpoint");
+
+$node->stop('immediate');
+$node->start;
+
+is( $node->safe_psql(
+ 'postgres',
+ qq{
+SELECT NOT active, invalidation_reason IS NULL,
+ restart_lsn = '$restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'target_slot'
+}),
+ 't|t|t',
+ 'target slot restores with its original restart LSN');
+ok(-f $restart_segment_path,
+ "target slot WAL segment $restart_segment exists after restart");
+
+$node->stop;
+
+done_testing();
diff --git a/src/test/recovery/t/058_slot_invalidation_concurrency.pl b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
new file mode 100644
index 00000000000..1259d7be425
--- /dev/null
+++ b/src/test/recovery/t/058_slot_invalidation_concurrency.pl
@@ -0,0 +1,273 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+#
+# Test concurrent invalidation of the same replication slot.
+#
+use strict;
+use warnings FATAL => 'all';
+
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Time::HiRes qw(usleep);
+
+use Test::More;
+
+if ($ENV{enable_injection_points} ne 'yes')
+{
+ plan skip_all => 'Injection points not supported by this build';
+}
+
+my $primary = PostgreSQL::Test::Cluster->new('primary');
+$primary->init(allows_streaming => 1, extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+wal_level = logical
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+
+if (!$primary->check_extension('injection_points'))
+{
+ plan skip_all => 'Extension injection_points not installed';
+}
+
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('phys')});
+$primary->backup('backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('standby');
+$standby->init_from_backup($primary, 'backup', has_streaming => 1);
+$standby->append_conf(
+ 'postgresql.conf', qq(
+primary_slot_name = 'phys'
+hot_standby_feedback = on
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+max_slot_wal_keep_size = 1MB
+log_checkpoints = on
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+my $injection_point = 'replication-slot-save-error';
+
+sub set_primary_wal_level
+{
+ my ($wal_level) = @_;
+
+ $primary->append_conf('postgresql.conf', "wal_level = $wal_level");
+ $primary->restart;
+}
+
+sub create_lagging_slot
+{
+ my ($slot_name) = @_;
+
+ $standby->create_logical_slot_on_standby($primary, $slot_name,
+ 'postgres');
+ $primary->advance_wal(8);
+ $primary->safe_psql('postgres', 'CHECKPOINT');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT injection_points_attach(
+ '$injection_point', 'wait', '$slot_name')
+});
+}
+
+sub backend_pid
+{
+ my ($backend_type) = @_;
+
+ return $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT pid
+FROM pg_stat_activity
+WHERE backend_type = '$backend_type'
+});
+}
+
+sub start_restartpoint
+{
+ my $checkpoint =
+ $standby->background_psql('postgres', on_error_stop => 0);
+
+ $checkpoint->set_query_timer_restart();
+ $checkpoint->query_until(
+ qr/checkpoint started/,
+ q(\echo checkpoint started
+CHECKPOINT;
+));
+
+ return $checkpoint;
+}
+
+sub finish_restartpoint
+{
+ my ($checkpoint) = @_;
+ my (undef, $error) = $checkpoint->query('SELECT 1', verbose => 0);
+
+ is($error, 0, 'restartpoint succeeds');
+ $checkpoint->quit;
+}
+
+sub wait_for_replication_slot_io
+{
+ my ($pid) = @_;
+
+ $standby->poll_query_until(
+ 'postgres',
+ qq{
+SELECT wait_event = 'ReplicationSlotIO'
+FROM pg_stat_activity
+WHERE pid = $pid
+}) or die "process $pid did not wait for replication slot I/O";
+}
+
+sub wake_invalidator
+{
+ my ($slot_name) = @_;
+ my $invalidation_reason;
+
+ foreach (1 .. 10 * $PostgreSQL::Test::Utils::timeout_default)
+ {
+ my $waiting = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT count(*) > 0
+FROM pg_stat_activity
+WHERE wait_event = '$injection_point'
+});
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')})
+ if $waiting eq 't';
+
+ $invalidation_reason = $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+});
+
+ last if $invalidation_reason ne '' && $waiting eq 'f';
+ usleep(100_000);
+ }
+
+ die "timed out waiting for slot $slot_name to be invalidated"
+ if !defined($invalidation_reason) || $invalidation_reason eq '';
+
+ return $invalidation_reason;
+}
+
+# The checkpointer starts invalidation before the startup process.
+my $slot_name = 'checkpointer_first';
+create_lagging_slot($slot_name);
+my $startup_pid = backend_pid('startup');
+my $checkpointer_pid = backend_pid('checkpointer');
+my $log_start = -s $standby->logfile;
+
+my $checkpoint = start_restartpoint();
+$standby->wait_for_event('checkpointer', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $checkpointer_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'checkpointer owns the slot while persisting invalidation');
+
+set_primary_wal_level('replica');
+wait_for_replication_slot_io($startup_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $checkpointer_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'startup process does not terminate the checkpointer');
+
+my $invalidation_reason = wake_invalidator($slot_name);
+finish_restartpoint($checkpoint);
+
+is($invalidation_reason, 'wal_removed',
+ 'slot is invalidated by the checkpointer');
+ok( !$standby->log_contains(
+ qr/canceling statement due to conflict with recovery/, $log_start),
+ 'checkpointer does not receive a recovery conflict');
+
+$primary->wait_for_replay_catchup($standby);
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+# The startup process starts invalidation before the checkpointer.
+set_primary_wal_level('logical');
+$primary->wait_for_replay_catchup($standby);
+
+$slot_name = 'startup_first';
+create_lagging_slot($slot_name);
+$startup_pid = backend_pid('startup');
+$checkpointer_pid = backend_pid('checkpointer');
+$log_start = -s $standby->logfile;
+
+set_primary_wal_level('replica');
+$standby->wait_for_event('startup', $injection_point);
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT active_pid = $startup_pid
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 't',
+ 'startup process owns the slot while persisting invalidation');
+
+$checkpoint = start_restartpoint();
+wait_for_replication_slot_io($checkpointer_pid);
+
+ok( !$standby->log_contains(
+ qr/terminating process $startup_pid to release replication slot
+ \s+"$slot_name"/x,
+ $log_start),
+ 'checkpointer does not terminate the startup process');
+
+$standby->safe_psql('postgres',
+ qq{SELECT injection_points_wakeup('$injection_point')});
+finish_restartpoint($checkpoint);
+$standby->wait_for_log(
+ qr/invalidating obsolete replication slot "$slot_name"/, $log_start);
+
+$primary->advance_wal(1);
+$primary->wait_for_replay_catchup($standby);
+is(backend_pid('startup'), $startup_pid, 'startup process survives');
+ok($standby->is_alive, 'standby remains running');
+
+if ($standby->is_alive)
+{
+ is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = '$slot_name'
+}),
+ 'wal_level_insufficient',
+ 'slot is invalidated by the startup process');
+
+ $standby->safe_psql('postgres',
+ qq{SELECT injection_points_detach('$injection_point')});
+
+ $standby->stop;
+}
+
+$primary->stop;
+
+done_testing();
--
2.34.1
[text/x-diff] v6-0002-Persist-synchronized-slot-invalidations-before-pu.patch (7.5K, ../../arTwLinbxvVkEqu2@bdtpg/3-v6-0002-Persist-synchronized-slot-invalidations-before-pu.patch)
download | inline diff:
From 0e6094ee5b32b80a66999c5774fd35ad8a846cfd Mon Sep 17 00:00:00 2001
From: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Date: Wed, 26 Aug 2026 05:36:40 +0000
Subject: [PATCH v6 2/2] Persist synchronized slot invalidations before
publishing them
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
synchronize_one_slot() publishes a remote slot's invalidation before saving
the local synchronized slot. A save failure therefore leaves the shared slot
invalid while its disk image remains valid. On the next synchronization,
drop_local_obsolete_slots() retains the local slot because the remote slot is
also invalidated. However, synchronize_one_slot() sees the local slot already
invalidated and skips the save, so the failed save is not retried directly.
Use ReplicationSlotPersistInvalidation() so the invalidated image is durable
before publication. Preserve the local restart LSN and recompute resource
horizons only after the invalidation becomes durable and visible.
A failed save now leaves the local slot valid, allowing the next synchronization
to retry. Add a primary and standby test covering the failure, a direct retry,
and an immediate restart that verifies durable invalidation.
Author: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
Reviewed-by: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
Reviewed-by: Miłosz Bieniek <bieniek.milosz@proton.me>
Reviewed-by: JoongHyuk Shin <sjh910805@gmail.com>
Reviewed-by: Rui Zhao <zhaorui126@gmail.com>
Reviewed-by: shveta malik <shveta.malik@gmail.com>
Discussion: https://postgr.es/m/ao7u5I9OeIR72kGp%40bdtpg
Backpatch-through: 17
---
src/backend/replication/logical/slotsync.c | 13 +-
src/backend/replication/slot.c | 2 +-
.../t/057_replslot_invalidation_durability.pl | 135 ++++++++++++++++++
3 files changed, 142 insertions(+), 8 deletions(-)
12.8% src/backend/replication/logical/
84.7% src/test/recovery/t/
diff --git a/src/backend/replication/logical/slotsync.c b/src/backend/replication/logical/slotsync.c
index c0403893e23..8191a3c4892 100644
--- a/src/backend/replication/logical/slotsync.c
+++ b/src/backend/replication/logical/slotsync.c
@@ -829,13 +829,12 @@ synchronize_one_slot(RemoteSlot *remote_slot, Oid remote_dbid,
if (slot->data.invalidated == RS_INVAL_NONE &&
remote_slot->invalidated != RS_INVAL_NONE)
{
- SpinLockAcquire(&slot->mutex);
- slot->data.invalidated = remote_slot->invalidated;
- SpinLockRelease(&slot->mutex);
-
- /* Make sure the invalidated state persists across server restart */
- ReplicationSlotMarkDirty();
- ReplicationSlotSave();
+ LWLockAcquire(&slot->io_in_progress_lock, LW_EXCLUSIVE);
+ ReplicationSlotPersistInvalidation(remote_slot->invalidated, false,
+ true);
+ LWLockRelease(&slot->io_in_progress_lock);
+ ReplicationSlotsComputeRequiredXmin(false);
+ ReplicationSlotsComputeRequiredLSN();
slot_updated = true;
}
diff --git a/src/backend/replication/slot.c b/src/backend/replication/slot.c
index ef54833e0f9..a645a9f7add 100644
--- a/src/backend/replication/slot.c
+++ b/src/backend/replication/slot.c
@@ -1204,7 +1204,7 @@ ReplicationSlotPersistInvalidation(ReplicationSlotInvalidationCause cause,
ReplicationSlot *slot = MyReplicationSlot;
Assert(slot != NULL);
- Assert(slot->data.persistency == RS_PERSISTENT);
+ Assert(slot->data.persistency != RS_EPHEMERAL);
Assert(slot->data.invalidated == RS_INVAL_NONE);
Assert(cause != RS_INVAL_NONE);
Assert(!clear_restart_lsn || cause == RS_INVAL_WAL_REMOVED);
diff --git a/src/test/recovery/t/057_replslot_invalidation_durability.pl b/src/test/recovery/t/057_replslot_invalidation_durability.pl
index 247d7b00dc1..483f038b767 100644
--- a/src/test/recovery/t/057_replslot_invalidation_durability.pl
+++ b/src/test/recovery/t/057_replslot_invalidation_durability.pl
@@ -122,4 +122,139 @@ ok(-f $restart_segment_path,
$node->stop;
+# Check that slot synchronization also persists an invalidation before
+# publishing it.
+my $primary = PostgreSQL::Test::Cluster->new('sync_primary');
+$primary->init(allows_streaming => 'logical', extra => ['--wal-segsize=1']);
+$primary->append_conf(
+ 'postgresql.conf', qq(
+autovacuum = off
+checkpoint_timeout = 1h
+max_wal_size = 64MB
+));
+$primary->start;
+$primary->safe_psql('postgres', 'CREATE EXTENSION injection_points');
+$primary->safe_psql('postgres',
+ q{SELECT pg_create_physical_replication_slot('sync_phys')});
+$primary->backup('sync_backup');
+
+my $standby = PostgreSQL::Test::Cluster->new('sync_standby');
+$standby->init_from_backup(
+ $primary, 'sync_backup',
+ has_streaming => 1,
+ has_restoring => 1);
+my $primary_connstr = $primary->connstr;
+$standby->append_conf(
+ 'postgresql.conf', qq(
+checkpoint_timeout = 1h
+hot_standby_feedback = on
+primary_slot_name = 'sync_phys'
+primary_conninfo = '$primary_connstr dbname=postgres'
+));
+$standby->start;
+$primary->wait_for_replay_catchup($standby);
+
+$primary->safe_psql(
+ 'postgres',
+ q{SELECT pg_create_logical_replication_slot(
+ 'sync_slot', 'pgoutput', false, false, true)});
+
+my $slot_synced = 'f';
+foreach (1 .. 10)
+{
+ $primary->safe_psql('postgres', 'SELECT pg_log_standby_snapshot()');
+ $primary->wait_for_replay_catchup($standby);
+ $standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+ $slot_synced = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT count(*) = 1
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+ AND synced
+ AND NOT temporary
+ AND invalidation_reason IS NULL
+});
+ last if $slot_synced eq 't';
+}
+is($slot_synced, 't', 'valid failover slot is synchronized');
+my $sync_restart_lsn = $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT restart_lsn
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+});
+
+$primary->append_conf('postgresql.conf', 'max_slot_wal_keep_size = 1MB');
+$primary->reload;
+$primary->advance_wal(8);
+$primary->wait_for_replay_catchup($standby);
+$primary->safe_psql('postgres', 'CHECKPOINT');
+
+is( $primary->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed',
+ 'failover slot is invalidated on the primary');
+
+$standby->safe_psql(
+ 'postgres',
+ q{
+SELECT injection_points_attach(
+ 'replication-slot-save-error', 'error', 'sync_slot')
+});
+
+($ret, $stdout, $stderr) =
+ $standby->psql('postgres', 'SELECT pg_sync_replication_slots()');
+like(
+ $stderr,
+ qr/error triggered for injection point replication-slot-save-error/,
+ 'injected error prevents synchronized invalidation from being saved');
+
+is( $standby->safe_psql(
+ 'postgres',
+ q{
+SELECT invalidation_reason IS NULL
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 't',
+ 'failed save leaves synchronized slot valid');
+
+$standby->safe_psql('postgres',
+ q{SELECT injection_points_detach('replication-slot-save-error')});
+
+$standby->safe_psql('postgres', 'SELECT pg_sync_replication_slots()');
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation is published');
+
+$standby->stop('immediate');
+$standby->start;
+
+is( $standby->safe_psql(
+ 'postgres',
+ qq{
+SELECT invalidation_reason, restart_lsn = '$sync_restart_lsn'
+FROM pg_replication_slots
+WHERE slot_name = 'sync_slot'
+}),
+ 'wal_removed|t',
+ 'retried synchronized invalidation and restart LSN survive restart');
+
+$standby->stop;
+$primary->stop;
+
done_testing();
--
2.34.1
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-25 04:08 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
1 sibling, 1 reply; 32+ messages in thread
From: shveta malik @ 2026-09-25 04:08 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Thu, Sep 24, 2026 at 3:11 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi,
>
> On Thu, Sep 24, 2026 at 10:55:03AM +0530, shveta malik wrote:
> > I had a look at patch001 as well. I have 2 questions:
> >
> > 1)
> > Would it be better to use a PG_TRY/PG_CATCH block in patch 001,
> > similar to patch 002? Currently, the slot is released via
> > PG_ENSURE_ERROR_CLEANUP, while we rely on top-level error cleanup for
> > lock-release. Using TRY/CATCH would let us explicitly release both the
> > slot and I/O lock together, making the error handling consistent
> > across both patches. We could even reuse persist_slot_invalidation()
> > with a small change to pass update_inactive_since from the caller.
>
> Yeah, makes sense. I changed 0001 to use PG_TRY/PG_CATCH and moved the common
> cleanup into ReplicationSlotPersistInvalidation(), with update_inactive_since
> passed by the caller.
>
> The I/O lock is still acquired by each caller because 0001 must hold it before
> claiming the inactive slot to serialize concurrent internal invalidators.
Yes, makes sense.
> > 2)
> > + /* Let caller know */
> > + invalidated = true;
> > + LWLockRelease(&s->io_in_progress_lock);
> > ReplicationSlotRelease();
> >
> > Wouldn't it be better (and safer) to release the slot before releasing
> > the I/O lock?
> >
> > Currently, concurrent invalidators are protected by the
> > 'invalidation_cause == RS_INVAL_NONE' check after acquiring the lock.
> > But releasing the slot first would close this race window entirely. It
> > would also make the order consistent with Patch 002 and the
> > error-handling flow in Patch 001 itself.
>
> The current ordering should be safe because the invalidation has already been
> published, so a concurrent invalidator exits before considering active_proc.
Oh I see. I missed this point earlier.
> That said, I agree that releasing the slot first means this ordering no longer
> relies on that check and makes the success and error paths consistent. So, done
> in the attached.
>
> It also adds the check you suggested for the synchronized slot's shared memory
> state after a successful synchronization.
>
Thanks for addressing comments. A few concerns on 001:
1)
In SaveSlotToPath(), should we add an 'Assert(cp.slotdata.restart_lsn
== InvalidXLogRecPtr)' at the end for the 'if (clear_restart_lsn)'
case?
slot->data.invalidated = invalidation_cause;
if (clear_restart_lsn)
+ {
+ Assert(cp.slotdata.restart_lsn == InvalidXLogRecPtr);
slot->data.restart_lsn = InvalidXLogRecPtr;
+ }
While slot->last_saved_restart_lsn correctly inherits
cp.slotdata.restart_lsn on the next line, adding this Assert
guarantees that the removed logic from
InvalidatePossiblyObsoleteSlot() was successfully compensated for in
the on-disk struct before we propagate it to shared memory. It is not
mandatory, but it would be good to have.
2)
+ Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
In ReplicationSlotReleaseInternal(), I didn’t quite understand the
reasoning behind above Assert. Does this mean that when the caller
passes update_inactive_since=true, the slot can even be temporary,
whereas if we are not updating inactive_since, the slot must be
persistent?
Is this based on the fact that slotsync passes update_inactive_since =
true and can have a temporary slot here? If so, that is not very clear
from the Assert itself. If we want to retain this check, should we
instead move the 'update_inactive_since ||' part to patch 002? IIUC,
patch 001 seems to strictly require RS_PERSISTENT. Let me know if I
understand it wrong.
I’m not sure what else we can do to make this clearer, but I think at
least a comment explaining why a temporary slot is allowed in this
case (update_inactive_since=true) would help.
3)
Another doubt I have is that with above Assert, when
update_inactive_since is TRUE, we are even allowing RS_EPHEMERAL
slots. However, ReplicationSlotPersistInvalidation() explicitly
disallows them in patch002 with:
Assert(slot->data.persistency != RS_EPHEMERAL);
Both checks are not in sync.
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-25 05:54 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: shveta malik <shveta.malik@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-25 05:54 UTC (permalink / raw)
To: shveta malik <shveta.malik@gmail.com>; +Cc: JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Fri, Sep 25, 2026 at 09:38:27AM +0530, shveta malik wrote:
> On Thu, Sep 24, 2026 at 3:11 PM Bertrand Drouvot
> <bertranddrouvot.pg@gmail.com> wrote:
> >
>
> Thanks for addressing comments. A few concerns on 001:
Thanks for looking at it!
> 1)
>
> In SaveSlotToPath(), should we add an 'Assert(cp.slotdata.restart_lsn
> == InvalidXLogRecPtr)' at the end for the 'if (clear_restart_lsn)'
> case?
>
> slot->data.invalidated = invalidation_cause;
> if (clear_restart_lsn)
> + {
> + Assert(cp.slotdata.restart_lsn == InvalidXLogRecPtr);
> slot->data.restart_lsn = InvalidXLogRecPtr;
> + }
>
> While slot->last_saved_restart_lsn correctly inherits
> cp.slotdata.restart_lsn on the next line, adding this Assert
> guarantees that the removed logic from
> InvalidatePossiblyObsoleteSlot() was successfully compensated for in
> the on-disk struct before we propagate it to shared memory. It is not
> mandatory, but it would be good to have.
I’m not sure this assertion adds much, since cp.slotdata.restart_lsn is explicitly
cleared above and is not modified afterward.
> 2)
> + Assert(update_inactive_since || slot->data.persistency == RS_PERSISTENT);
>
> In ReplicationSlotReleaseInternal(), I didn’t quite understand the
> reasoning behind above Assert. Does this mean that when the caller
> passes update_inactive_since=true, the slot can even be temporary,
> whereas if we are not updating inactive_since, the slot must be
> persistent?
Yes. In fact, with update_inactive_since=true it can also be ephemeral, since
ReplicationSlotRelease() uses that value for the ordinary release path.
This is not specific to slotsync. The false case is introduced by 0001 and is
only used to preserve inactive_since when rolling back ownership of an inactive
persistent slot.
Maybe the following comment would make that clearer?
"
/*
* Skipping the inactive_since update is only needed when undoing the
* internal acquisition of an inactive persistent slot after an ERROR.
*/
"
> 3)
> Another doubt I have is that with above Assert, when
> update_inactive_since is TRUE, we are even allowing RS_EPHEMERAL
> slots. However, ReplicationSlotPersistInvalidation() explicitly
> disallows them in patch002 with:
>
> Assert(slot->data.persistency != RS_EPHEMERAL);
>
> Both checks are not in sync.
I think they apply to different scopes. ReplicationSlotReleaseInternal() is the
general release implementation, so update_inactive_since=true imposes no
persistency restriction. In particular, an ephemeral slot is dropped by that
path.
ReplicationSlotPersistInvalidation() has a narrower contract and is only
intended for persistent or temporary slots. That said, maybe its Assert could
express all the supported combinations more clearly?
"
Assert(slot->data.persistency == RS_PERSISTENT ||
(slot->data.persistency == RS_TEMPORARY &&
update_inactive_since));
"
Regards,
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-25 16:18 Rui Zhao <zhaorui126@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: Rui Zhao @ 2026-09-25 16:18 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: shveta malik <shveta.malik@gmail.com>; JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; pgsql-hackers@lists.postgresql.org
On 2026-Sep-25 at 05:54 UTC, Bertrand Drouvot wrote:
> ReplicationSlotPersistInvalidation() has a narrower contract and is only
> intended for persistent or temporary slots.
Since this function also handles temporary slots, could we qualify
the following sentence in 0002's commit message?
> A failed save now leaves the local slot valid, allowing the next
> synchronization to retry.
Could we change that to:
A failed save now leaves a persistent local slot valid, allowing the
next synchronization to retry.
I suggest adding "persistent" because SQL error cleanup deletes
temporary synchronized slots. In my v6 test:
1. I ran pg_sync_replication_slots() on the standby. The local
pending_slot remained temporary because the primary slot's
restart_lsn was behind the standby slot's restart_lsn.
2. While that call was still running, I made
pg_replslot/pending_slot/state.tmp a directory on the standby, then
invalidated pending_slot on the primary (wal_removed).
Synchronization then tried to save that invalidation on the standby.
The save failed with "File exists", and error cleanup deleted the
temporary slot.
3. I called pg_sync_replication_slots() again in the same connection.
It completed without error but did not recreate pending_slot because
the primary slot was already invalidated.
Regards,
Rui
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-27 14:10 Zhijie Hou <houzhijie22@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
1 sibling, 1 reply; 32+ messages in thread
From: Zhijie Hou @ 2026-09-27 14:10 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: shveta malik <shveta.malik@gmail.com>; JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Thu, Sep 24, 2026 at 5:41 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
> On Thu, Sep 24, 2026 at 10:55:03AM +0530, shveta malik wrote:
> > 2)
> > + /* Let caller know */
> > + invalidated = true;
> > + LWLockRelease(&s->io_in_progress_lock);
> > ReplicationSlotRelease();
> >
> > Wouldn't it be better (and safer) to release the slot before releasing
> > the I/O lock?
> >
> > Currently, concurrent invalidators are protected by the
> > 'invalidation_cause == RS_INVAL_NONE' check after acquiring the lock.
> > But releasing the slot first would close this race window entirely. It
> > would also make the order consistent with Patch 002 and the
> > error-handling flow in Patch 001 itself.
>
> The current ordering should be safe because the invalidation has already been
> published, so a concurrent invalidator exits before considering active_proc.
>
> That said, I agree that releasing the slot first means this ordering no longer
> relies on that check and makes the success and error paths consistent. So, done
> in the attached.
>
> It also adds the check you suggested for the synchronized slot's shared memory
> state after a successful synchronization.
Thanks for updating the patches.
When reading the patches, the part that feels heavy to me is the serialization
machinery added to InvalidatePossiblyObsoleteSlot() for the two-invalidator
race - the conditional acquire of io_in_progress_lock, dropping
ReplicationSlotControlLock to wait, and the restart of the loop - plus the
caller-owns-the-io-lock contract that ReplicationSlotPersistInvalidation()
imposes on both call sites. I understand why it's needed once claim and publish
stop being atomic (the SIGTERM that kills the startup process and shuts down
the standby is nasty), but I wonder if we can avoid making them non-atomic in
the first place.
You mentioned effective_catalog_xmin, and there are similar shadow fields like
last_saved_restart_lsn. What about the same style here: keep the claim exactly
as on master - active_proc and data.invalidated set in one spinlock section -
and add a pure in-memory boolean, say invalidation_durable, set only at the
point the invalid image has actually been written and fsynced (the tail of
SaveSlotToPath(), keyed off the image just written. All consumer references to
data.invalidated (horizon computations, pg_replication_slots, slotsync's
skip/drop decisions) would consult the new flag instead; the invalidators'
mutual-exclusion check and the acquire path keep reading the cause as today.
The new flag can be added to the padding space, so there is no change in the
size of ReplicationSlot.
I'm not insisting that we change the approach, just wanted to share an
alternative for discussion.
Best Regards,
Zhijie Hou
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-28 07:46 Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
parent: Zhijie Hou <houzhijie22@gmail.com>
0 siblings, 1 reply; 32+ messages in thread
From: Bertrand Drouvot @ 2026-09-28 07:46 UTC (permalink / raw)
To: Zhijie Hou <houzhijie22@gmail.com>; +Cc: shveta malik <shveta.malik@gmail.com>; JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi,
On Sun, Sep 27, 2026 at 10:10:33PM +0800, Zhijie Hou wrote:
> When reading the patches,
Thanks for looking at it!
> the part that feels heavy to me is the serialization
> machinery added to InvalidatePossiblyObsoleteSlot() for the two-invalidator
> race - the conditional acquire of io_in_progress_lock, dropping
> ReplicationSlotControlLock to wait, and the restart of the loop - plus the
> caller-owns-the-io-lock contract that ReplicationSlotPersistInvalidation()
> imposes on both call sites.
That might look heavy but I don't think this pattern is unusual: SLRU uses the
same general lock, wait, and recheck pattern, and InvalidatePossiblyObsoleteSlot()
already follows that model when waiting on active_cv.
> You mentioned effective_catalog_xmin, and there are similar shadow fields like
> last_saved_restart_lsn. What about the same style here: keep the claim exactly
> as on master - active_proc and data.invalidated set in one spinlock section -
> and add a pure in-memory boolean, say invalidation_durable, set only at the
> point the invalid image has actually been written and fsynced (the tail of
> SaveSlotToPath(), keyed off the image just written. All consumer references to
> data.invalidated (horizon computations, pg_replication_slots, slotsync's
> skip/drop decisions) would consult the new flag instead; the invalidators'
> mutual-exclusion check and the acquire path keep reading the cause as today.
I'm not sure the alternative is lighter overall. It moves the complexity into
a new intermediate slot state and requires each consumer of data.invalidated to
decide whether it should also check invalidation_durable.
> The new flag can be added to the padding space, so there is no change in the
> size of ReplicationSlot.
Yeah that look ok if, for example, we place it here:
(gdb) ptype /o struct ReplicationSlot
/* offset | size */ type = struct ReplicationSlot {
/* 0 | 1 */ slock_t mutex;
/* 1 | 1 */ _Bool in_use;
/* XXX 2-byte hole */
/* 4 | 4 */ ProcNumber active_proc;
/* 8 | 1 */ _Bool just_dirtied;
/* 9 | 1 */ _Bool dirty;
/* XXX 2-byte hole */
/* 12 | 4 */ TransactionId effective_xmin;
.
.
.
My concern is that data.invalidated would then have two roles depending on
invalidation_durable. Invalidators and the acquisition path would treat a value
other than RS_INVAL_NONE as an invalidation, while other consumers would do so
only once invalidation_durable is set.
That could be an issue for existing extensions on back branches. An extension
could treat the slot as invalid while core consumers gated by invalidation_durable
still treat the invalidation as not effective. So, although the ABI layout would
be preserved, the semantics of an existing field would change.
Thoughts?
--
Bertrand Drouvot
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com
^ permalink raw reply [nested|flat] 32+ messages in thread
* Re: Persist slot invalidations before publishing them
@ 2026-09-28 10:33 shveta malik <shveta.malik@gmail.com>
parent: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
0 siblings, 0 replies; 32+ messages in thread
From: shveta malik @ 2026-09-28 10:33 UTC (permalink / raw)
To: Bertrand Drouvot <bertranddrouvot.pg@gmail.com>; +Cc: Zhijie Hou <houzhijie22@gmail.com>; JoongHyuk Shin <sjh910805@gmail.com>; Amit Kapila <amit.kapila16@gmail.com>; Rui Zhao <zhaorui126@gmail.com>; pgsql-hackers@lists.postgresql.org, shveta malik <shveta.malik@gmail.com>
On Mon, Sep 28, 2026 at 1:16 PM Bertrand Drouvot
<bertranddrouvot.pg@gmail.com> wrote:
>
> Hi,
>
> On Sun, Sep 27, 2026 at 10:10:33PM +0800, Zhijie Hou wrote:
> > When reading the patches,
>
> Thanks for looking at it!
>
> > the part that feels heavy to me is the serialization
> > machinery added to InvalidatePossiblyObsoleteSlot() for the two-invalidator
> > race - the conditional acquire of io_in_progress_lock, dropping
> > ReplicationSlotControlLock to wait, and the restart of the loop - plus the
> > caller-owns-the-io-lock contract that ReplicationSlotPersistInvalidation()
> > imposes on both call sites.
>
> That might look heavy but I don't think this pattern is unusual: SLRU uses the
> same general lock, wait, and recheck pattern, and InvalidatePossiblyObsoleteSlot()
> already follows that model when waiting on active_cv.
>
> > You mentioned effective_catalog_xmin, and there are similar shadow fields like
> > last_saved_restart_lsn. What about the same style here: keep the claim exactly
> > as on master - active_proc and data.invalidated set in one spinlock section -
> > and add a pure in-memory boolean, say invalidation_durable, set only at the
> > point the invalid image has actually been written and fsynced (the tail of
> > SaveSlotToPath(), keyed off the image just written. All consumer references to
> > data.invalidated (horizon computations, pg_replication_slots, slotsync's
> > skip/drop decisions) would consult the new flag instead; the invalidators'
> > mutual-exclusion check and the acquire path keep reading the cause as today.
>
> I'm not sure the alternative is lighter overall. It moves the complexity into
> a new intermediate slot state and requires each consumer of data.invalidated to
> decide whether it should also check invalidation_durable.
>
> > The new flag can be added to the padding space, so there is no change in the
> > size of ReplicationSlot.
>
> Yeah that look ok if, for example, we place it here:
>
> (gdb) ptype /o struct ReplicationSlot
> /* offset | size */ type = struct ReplicationSlot {
> /* 0 | 1 */ slock_t mutex;
> /* 1 | 1 */ _Bool in_use;
> /* XXX 2-byte hole */
> /* 4 | 4 */ ProcNumber active_proc;
> /* 8 | 1 */ _Bool just_dirtied;
> /* 9 | 1 */ _Bool dirty;
> /* XXX 2-byte hole */
> /* 12 | 4 */ TransactionId effective_xmin;
> .
> .
> .
>
> My concern is that data.invalidated would then have two roles depending on
> invalidation_durable. Invalidators and the acquisition path would treat a value
> other than RS_INVAL_NONE as an invalidation, while other consumers would do so
> only once invalidation_durable is set.
>
> That could be an issue for existing extensions on back branches. An extension
> could treat the slot as invalid while core consumers gated by invalidation_durable
> still treat the invalidation as not effective. So, although the ABI layout would
> be preserved, the semantics of an existing field would change.
>
> Thoughts?
I think the new approach could make the code significantly more
fragile and could lead to silent system corruption if not handled
carefully. For example, we could hit a case where logical decoding is
disabled on seeing the last invalidated slot and WALs are removed, but
then the fsync to disk fails. This would make the system completely
unrecoverable (system startup would fail because logical decoding was
disabled while a valid slot is still present on disk). Although we
will theoretically change all readers of invalidated to also consult
the shadow flag, my point is that a single missed check would result
in an unrecoverable state.
And not just existing extensions but even future ones could easily
fall into this exact trap by missing just one check.
thanks
Shveta
^ permalink raw reply [nested|flat] 32+ messages in thread
end of thread, other threads:[~2026-09-28 10:33 UTC | newest]
Thread overview: 32+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2026-08-26 05:34 [PATCH v1 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 05:34 [PATCH v2 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 05:34 [PATCH v3 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 05:34 [PATCH v4 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 05:34 [PATCH v5 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 05:34 [PATCH v6 1/2] Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 13:49 Persist slot invalidations before publishing them Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-26 19:24 ` Miłosz Bieniek <bieniek.milosz@proton.me>
2026-08-27 01:30 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-27 08:06 ` Kyotaro Horiguchi <horikyota.ntt@gmail.com>
2026-08-27 10:26 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-08-27 11:12 ` Osama Abdul Qader <osamaabdulqader.cs@gmail.com>
2026-08-28 10:29 ` Amit Kapila <amit.kapila16@gmail.com>
2026-08-28 13:32 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-13 10:31 ` JoongHyuk Shin <sjh910805@gmail.com>
2026-09-22 08:27 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-22 10:16 ` shveta malik <shveta.malik@gmail.com>
2026-09-23 06:06 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-23 06:30 ` shveta malik <shveta.malik@gmail.com>
2026-09-23 08:38 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-23 10:37 ` shveta malik <shveta.malik@gmail.com>
2026-09-23 16:05 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-24 03:49 ` shveta malik <shveta.malik@gmail.com>
2026-09-24 05:25 ` shveta malik <shveta.malik@gmail.com>
2026-09-24 09:41 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-25 04:08 ` shveta malik <shveta.malik@gmail.com>
2026-09-25 05:54 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-25 16:18 ` Rui Zhao <zhaorui126@gmail.com>
2026-09-27 14:10 ` Zhijie Hou <houzhijie22@gmail.com>
2026-09-28 07:46 ` Bertrand Drouvot <bertranddrouvot.pg@gmail.com>
2026-09-28 10:33 ` shveta malik <shveta.malik@gmail.com>
2026-09-17 06:51 ` Rui Zhao <zhaorui126@gmail.com>
This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox