agora inbox for pgsql-hackers@postgresql.org
help / color / mirror / Atom feed[PATCH 3/3] Initial steps to making relkind an enum
35+ messages / 11 participants
[nested] [flat]
* [PATCH 3/3] Initial steps to making relkind an enum
@ 2025-10-31 13:07 Álvaro Herrera <alvherre@kurilemu.de>
0 siblings, 0 replies; 35+ messages in thread
From: Álvaro Herrera @ 2025-10-31 13:07 UTC (permalink / raw)
---
contrib/pg_overexplain/pg_overexplain.c | 4 +---
contrib/pg_prewarm/pg_prewarm.c | 2 +-
contrib/postgres_fdw/postgres_fdw.c | 2 +-
contrib/sepgsql/dml.c | 2 +-
contrib/sepgsql/relation.c | 14 +++++------
src/backend/access/common/reloptions.c | 2 +-
src/backend/catalog/aclchk.c | 6 ++---
src/backend/catalog/heap.c | 14 +++++------
src/backend/catalog/index.c | 2 +-
src/backend/catalog/objectaddress.c | 2 +-
src/backend/catalog/pg_class.c | 2 +-
src/backend/catalog/pg_publication.c | 6 ++---
src/backend/catalog/pg_subscription.c | 4 ++--
src/backend/commands/copyto.c | 2 +-
src/backend/commands/createas.c | 2 +-
src/backend/commands/indexcmds.c | 12 +++++-----
src/backend/commands/lockcmds.c | 4 ++--
src/backend/commands/policy.c | 2 +-
src/backend/commands/publicationcmds.c | 2 +-
src/backend/commands/subscriptioncmds.c | 8 +++----
src/backend/commands/tablecmds.c | 16 ++++++-------
src/backend/commands/trigger.c | 2 +-
src/backend/executor/nodeModifyTable.c | 4 ++--
src/backend/nodes/gen_node_support.pl | 3 ++-
src/backend/optimizer/util/appendinfo.c | 2 +-
src/backend/replication/pgoutput/pgoutput.c | 2 +-
src/backend/statistics/stat_utils.c | 2 +-
src/backend/tcop/utility.c | 2 +-
src/backend/utils/activity/pgstat_relation.c | 2 +-
src/backend/utils/adt/acl.c | 6 ++---
src/backend/utils/adt/partitionfuncs.c | 4 ++--
src/backend/utils/adt/ruleutils.c | 2 +-
src/backend/utils/cache/lsyscache.c | 6 ++---
src/backend/utils/cache/relcache.c | 2 +-
src/bin/pg_dump/pg_backup_archiver.c | 2 +-
src/bin/pg_dump/pg_backup_archiver.h | 5 ++--
src/bin/pg_dump/pg_dump.c | 10 ++++----
src/bin/pg_dump/pg_dump.h | 6 ++---
src/bin/pgbench/pgbench.c | 2 +-
src/bin/psql/describe.c | 6 ++---
src/include/access/reloptions.h | 3 ++-
src/include/catalog/heap.h | 7 +++---
src/include/catalog/objectaddress.h | 2 +-
src/include/catalog/pg_class.h | 25 +++++++++++---------
src/include/catalog/pg_publication.h | 2 +-
src/include/commands/tablecmds.h | 2 +-
src/include/nodes/parsenodes.h | 3 ++-
src/include/replication/logicalproto.h | 2 +-
src/include/utils/lsyscache.h | 3 ++-
src/include/utils/relcache.h | 3 ++-
50 files changed, 120 insertions(+), 112 deletions(-)
diff --git a/contrib/pg_overexplain/pg_overexplain.c b/contrib/pg_overexplain/pg_overexplain.c
index 316ffd1c87f..d575d462d97 100644
--- a/contrib/pg_overexplain/pg_overexplain.c
+++ b/contrib/pg_overexplain/pg_overexplain.c
@@ -534,10 +534,8 @@ overexplain_range_table(PlannedStmt *plannedstmt, ExplainState *es)
case RELKIND_PARTITIONED_INDEX:
relkind = "partitioned_index";
break;
- case '\0':
- relkind = NULL;
- break;
default:
+ pg_unreachable();
relkind = psprintf("%c", rte->relkind);
break;
}
diff --git a/contrib/pg_prewarm/pg_prewarm.c b/contrib/pg_prewarm/pg_prewarm.c
index c2716086693..f320a0e3831 100644
--- a/contrib/pg_prewarm/pg_prewarm.c
+++ b/contrib/pg_prewarm/pg_prewarm.c
@@ -73,7 +73,7 @@ pg_prewarm(PG_FUNCTION_ARGS)
char *ttype;
PrewarmType ptype;
AclResult aclresult;
- char relkind;
+ Relkind relkind;
Oid privOid;
/* Basic sanity checking. */
diff --git a/contrib/postgres_fdw/postgres_fdw.c b/contrib/postgres_fdw/postgres_fdw.c
index e0457bb1253..5ff70f7ddfa 100644
--- a/contrib/postgres_fdw/postgres_fdw.c
+++ b/contrib/postgres_fdw/postgres_fdw.c
@@ -4951,7 +4951,7 @@ postgresGetAnalyzeInfoForForeignTable(Relation relation, bool *can_tablesample)
StringInfoData sql;
PGresult *res;
double reltuples;
- char relkind;
+ Relkind relkind;
/* assume the remote relation does not support TABLESAMPLE */
*can_tablesample = false;
diff --git a/contrib/sepgsql/dml.c b/contrib/sepgsql/dml.c
index b44d09e447f..4f5864b22da 100644
--- a/contrib/sepgsql/dml.c
+++ b/contrib/sepgsql/dml.c
@@ -150,7 +150,7 @@ check_relation_privileges(Oid relOid,
char *audit_name;
Bitmapset *columns;
int index;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
bool result = true;
/*
diff --git a/contrib/sepgsql/relation.c b/contrib/sepgsql/relation.c
index 5c61e1f4c81..a77a8b5a846 100644
--- a/contrib/sepgsql/relation.c
+++ b/contrib/sepgsql/relation.c
@@ -52,7 +52,7 @@ sepgsql_attribute_post_create(Oid relOid, AttrNumber attnum)
ObjectAddress object;
Form_pg_attribute attForm;
StringInfoData audit_name;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
/*
* Only attributes within regular relations or partition relations have
@@ -134,7 +134,7 @@ sepgsql_attribute_drop(Oid relOid, AttrNumber attnum)
{
ObjectAddress object;
char *audit_name;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
if (relkind != RELKIND_RELATION && relkind != RELKIND_PARTITIONED_TABLE)
return;
@@ -167,7 +167,7 @@ sepgsql_attribute_relabel(Oid relOid, AttrNumber attnum,
{
ObjectAddress object;
char *audit_name;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
if (relkind != RELKIND_RELATION && relkind != RELKIND_PARTITIONED_TABLE)
ereport(ERROR,
@@ -210,7 +210,7 @@ sepgsql_attribute_setattr(Oid relOid, AttrNumber attnum)
{
ObjectAddress object;
char *audit_name;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
if (relkind != RELKIND_RELATION && relkind != RELKIND_PARTITIONED_TABLE)
return;
@@ -418,7 +418,7 @@ sepgsql_relation_drop(Oid relOid)
ObjectAddress object;
char *audit_name;
uint16_t tclass = 0;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
switch (relkind)
{
@@ -526,7 +526,7 @@ sepgsql_relation_truncate(Oid relOid)
ObjectAddress object;
char *audit_name;
uint16_t tclass = 0;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
switch (relkind)
{
@@ -565,7 +565,7 @@ sepgsql_relation_relabel(Oid relOid, const char *seclabel)
{
ObjectAddress object;
char *audit_name;
- char relkind = get_rel_relkind(relOid);
+ Relkind relkind = get_rel_relkind(relOid);
uint16_t tclass = 0;
if (relkind == RELKIND_RELATION || relkind == RELKIND_PARTITIONED_TABLE)
diff --git a/src/backend/access/common/reloptions.c b/src/backend/access/common/reloptions.c
index 237ab8d0ed9..00b3588cded 100644
--- a/src/backend/access/common/reloptions.c
+++ b/src/backend/access/common/reloptions.c
@@ -2149,7 +2149,7 @@ view_reloptions(Datum reloptions, bool validate)
* Parse options for heaps, views and toast tables.
*/
bytea *
-heap_reloptions(char relkind, Datum reloptions, bool validate)
+heap_reloptions(Relkind relkind, Datum reloptions, bool validate)
{
StdRdOptions *rdopts;
diff --git a/src/backend/catalog/aclchk.c b/src/backend/catalog/aclchk.c
index a431fc0926f..18a43e15509 100644
--- a/src/backend/catalog/aclchk.c
+++ b/src/backend/catalog/aclchk.c
@@ -123,7 +123,7 @@ static void SetDefaultACL(InternalDefaultACL *iacls);
static List *objectNamesToOids(ObjectType objtype, List *objnames,
bool is_grant);
static List *objectsInSchemaToOids(ObjectType objtype, List *nspnames);
-static List *getRelationsInNamespace(Oid namespaceId, char relkind);
+static List *getRelationsInNamespace(Oid namespaceId, Relkind relkind);
static void expand_col_privileges(List *colnames, Oid table_oid,
AclMode this_privileges,
AclMode *col_privileges,
@@ -875,7 +875,7 @@ objectsInSchemaToOids(ObjectType objtype, List *nspnames)
* Return Oid list of relations in given namespace filtered by relation kind
*/
static List *
-getRelationsInNamespace(Oid namespaceId, char relkind)
+getRelationsInNamespace(Oid namespaceId, Relkind relkind)
{
List *relations = NIL;
ScanKeyData key[2];
@@ -890,7 +890,7 @@ getRelationsInNamespace(Oid namespaceId, char relkind)
ScanKeyInit(&key[1],
Anum_pg_class_relkind,
BTEqualStrategyNumber, F_CHAREQ,
- CharGetDatum(relkind));
+ CharGetDatum((char) relkind));
rel = table_open(RelationRelationId, AccessShareLock);
scan = table_beginscan_catalog(rel, 2, key);
diff --git a/src/backend/catalog/heap.c b/src/backend/catalog/heap.c
index 606434823cf..fa17893855d 100644
--- a/src/backend/catalog/heap.c
+++ b/src/backend/catalog/heap.c
@@ -89,7 +89,7 @@ static void AddNewRelationTuple(Relation pg_class_desc,
Oid new_type_oid,
Oid reloftype,
Oid relowner,
- char relkind,
+ Relkind relkind,
TransactionId relfrozenxid,
TransactionId relminmxid,
Datum relacl,
@@ -289,7 +289,7 @@ heap_create(const char *relname,
RelFileNumber relfilenumber,
Oid accessmtd,
TupleDesc tupDesc,
- char relkind,
+ Relkind relkind,
char relpersistence,
bool shared_relation,
bool mapped_relation,
@@ -449,7 +449,7 @@ heap_create(const char *relname,
* --------------------------------
*/
void
-CheckAttributeNamesTypes(TupleDesc tupdesc, char relkind,
+CheckAttributeNamesTypes(TupleDesc tupdesc, Relkind relkind,
int flags)
{
int i;
@@ -833,7 +833,7 @@ InsertPgAttributeTuples(Relation pg_attribute_rel,
static void
AddNewAttributeTuples(Oid new_rel_oid,
TupleDesc tupdesc,
- char relkind)
+ Relkind relkind)
{
Relation rel;
CatalogIndexState indstate;
@@ -939,7 +939,7 @@ InsertPgClassTuple(Relation pg_class_desc,
values[Anum_pg_class_relhasindex - 1] = BoolGetDatum(rd_rel->relhasindex);
values[Anum_pg_class_relisshared - 1] = BoolGetDatum(rd_rel->relisshared);
values[Anum_pg_class_relpersistence - 1] = CharGetDatum(rd_rel->relpersistence);
- values[Anum_pg_class_relkind - 1] = CharGetDatum(rd_rel->relkind);
+ values[Anum_pg_class_relkind - 1] = CharGetDatum((char) rd_rel->relkind);
values[Anum_pg_class_relnatts - 1] = Int16GetDatum(rd_rel->relnatts);
values[Anum_pg_class_relchecks - 1] = Int16GetDatum(rd_rel->relchecks);
values[Anum_pg_class_relhasrules - 1] = BoolGetDatum(rd_rel->relhasrules);
@@ -987,7 +987,7 @@ AddNewRelationTuple(Relation pg_class_desc,
Oid new_type_oid,
Oid reloftype,
Oid relowner,
- char relkind,
+ Relkind relkind,
TransactionId relfrozenxid,
TransactionId relminmxid,
Datum relacl,
@@ -1129,7 +1129,7 @@ heap_create_with_catalog(const char *relname,
Oid accessmtd,
TupleDesc tupdesc,
List *cooked_constraints,
- char relkind,
+ Relkind relkind,
char relpersistence,
bool shared_relation,
bool mapped_relation,
diff --git a/src/backend/catalog/index.c b/src/backend/catalog/index.c
index 43de42ce39e..ede4546ecee 100644
--- a/src/backend/catalog/index.c
+++ b/src/backend/catalog/index.c
@@ -757,7 +757,7 @@ index_create(Relation heapRelation,
bool invalid = (flags & INDEX_CREATE_INVALID) != 0;
bool concurrent = (flags & INDEX_CREATE_CONCURRENT) != 0;
bool partitioned = (flags & INDEX_CREATE_PARTITIONED) != 0;
- char relkind;
+ Relkind relkind;
TransactionId relfrozenxid;
MultiXactId relminmxid;
bool create_storage = !RelFileNumberIsValid(relFileNumber);
diff --git a/src/backend/catalog/objectaddress.c b/src/backend/catalog/objectaddress.c
index 02af64b82c6..93b6e26d067 100644
--- a/src/backend/catalog/objectaddress.c
+++ b/src/backend/catalog/objectaddress.c
@@ -6182,7 +6182,7 @@ strlist_to_textarray(List *list)
* message saying "table" than fail entirely.
*/
ObjectType
-get_relkind_objtype(char relkind)
+get_relkind_objtype(Relkind relkind)
{
switch (relkind)
{
diff --git a/src/backend/catalog/pg_class.c b/src/backend/catalog/pg_class.c
index 38cf89f09fa..a841ed20868 100644
--- a/src/backend/catalog/pg_class.c
+++ b/src/backend/catalog/pg_class.c
@@ -21,7 +21,7 @@
* operation.
*/
int
-errdetail_relkind_not_supported(char relkind)
+errdetail_relkind_not_supported(Relkind relkind)
{
switch (relkind)
{
diff --git a/src/backend/catalog/pg_publication.c b/src/backend/catalog/pg_publication.c
index 9a4791c573e..624ebd59697 100644
--- a/src/backend/catalog/pg_publication.c
+++ b/src/backend/catalog/pg_publication.c
@@ -866,7 +866,7 @@ GetAllTablesPublications(void)
* publication.
*/
List *
-GetAllPublicationRelations(char relkind, bool pubviaroot)
+GetAllPublicationRelations(Relkind relkind, bool pubviaroot)
{
Relation classRel;
ScanKeyData key[1];
@@ -881,7 +881,7 @@ GetAllPublicationRelations(char relkind, bool pubviaroot)
ScanKeyInit(&key[0],
Anum_pg_class_relkind,
BTEqualStrategyNumber, F_CHAREQ,
- CharGetDatum(relkind));
+ CharGetDatum((char) relkind));
scan = table_beginscan_catalog(classRel, 1, key);
@@ -1016,7 +1016,7 @@ GetSchemaPublicationRelations(Oid schemaid, PublicationPartOpt pub_partopt)
{
Form_pg_class relForm = (Form_pg_class) GETSTRUCT(tuple);
Oid relid = relForm->oid;
- char relkind;
+ Relkind relkind;
if (!is_publishable_class(relid, relForm))
continue;
diff --git a/src/backend/catalog/pg_subscription.c b/src/backend/catalog/pg_subscription.c
index 2b103245290..988e80c2c31 100644
--- a/src/backend/catalog/pg_subscription.c
+++ b/src/backend/catalog/pg_subscription.c
@@ -537,7 +537,7 @@ HasSubscriptionTables(Oid subid)
while (HeapTupleIsValid(tup = systable_getnext(scan)))
{
Form_pg_subscription_rel subrel;
- char relkind;
+ Relkind relkind;
subrel = (Form_pg_subscription_rel) GETSTRUCT(tup);
relkind = get_rel_relkind(subrel->srrelid);
@@ -600,7 +600,7 @@ GetSubscriptionRelations(Oid subid, bool tables, bool sequences,
SubscriptionRelState *relstate;
Datum d;
bool isnull;
- char relkind;
+ Relkind relkind;
subrel = (Form_pg_subscription_rel) GETSTRUCT(tup);
diff --git a/src/backend/commands/copyto.c b/src/backend/commands/copyto.c
index 4ab4a3893d5..8d7530ea03b 100644
--- a/src/backend/commands/copyto.c
+++ b/src/backend/commands/copyto.c
@@ -692,7 +692,7 @@ BeginCopyTo(ParseState *pstate,
foreach_oid(child, children)
{
- char relkind = get_rel_relkind(child);
+ Relkind relkind = get_rel_relkind(child);
if (relkind == RELKIND_FOREIGN_TABLE)
{
diff --git a/src/backend/commands/createas.c b/src/backend/commands/createas.c
index 270e9bf3110..b05161a9aab 100644
--- a/src/backend/commands/createas.c
+++ b/src/backend/commands/createas.c
@@ -83,7 +83,7 @@ create_ctas_internal(List *attrList, IntoClause *into)
{
CreateStmt *create = makeNode(CreateStmt);
bool is_matview;
- char relkind;
+ Relkind relkind;
Datum toast_options;
const char *const validnsps[] = HEAP_RELOPT_NAMESPACES;
ObjectAddress intoRelationAddr;
diff --git a/src/backend/commands/indexcmds.c b/src/backend/commands/indexcmds.c
index 635679cc1f2..dfb15f91931 100644
--- a/src/backend/commands/indexcmds.c
+++ b/src/backend/commands/indexcmds.c
@@ -133,7 +133,7 @@ typedef struct ReindexErrorInfo
{
char *relname;
char *relnamespace;
- char relkind;
+ Relkind relkind;
} ReindexErrorInfo;
/*
@@ -2947,7 +2947,7 @@ ReindexIndex(const ReindexStmt *stmt, const ReindexParams *params, bool isTopLev
struct ReindexIndexCallbackState state;
Oid indOid;
char persistence;
- char relkind;
+ Relkind relkind;
/*
* Find and lock index, and check permissions on table; use callback to
@@ -2998,7 +2998,7 @@ static void
RangeVarCallbackForReindexIndex(const RangeVar *relation,
Oid relId, Oid oldRelId, void *arg)
{
- char relkind;
+ Relkind relkind;
struct ReindexIndexCallbackState *state = arg;
LOCKMODE table_lockmode;
Oid table_oid;
@@ -3373,7 +3373,7 @@ static void
ReindexPartitions(const ReindexStmt *stmt, Oid relid, const ReindexParams *params, bool isTopLevel)
{
List *partitions = NIL;
- char relkind = get_rel_relkind(relid);
+ Relkind relkind = get_rel_relkind(relid);
char *relname = get_rel_name(relid);
char *relnamespace = get_namespace_name(get_rel_namespace(relid));
MemoryContext reindex_context;
@@ -3474,7 +3474,7 @@ ReindexMultipleInternal(const ReindexStmt *stmt, const List *relids, const Reind
foreach(l, relids)
{
Oid relid = lfirst_oid(l);
- char relkind;
+ Relkind relkind;
char relpersistence;
StartTransactionCommand();
@@ -3608,7 +3608,7 @@ ReindexRelationConcurrently(const ReindexStmt *stmt, Oid relationOid, const Rein
*lc2;
MemoryContext private_context;
MemoryContext oldcontext;
- char relkind;
+ Relkind relkind;
char *relationName = NULL;
char *relationNamespace = NULL;
PGRUsage ru0;
diff --git a/src/backend/commands/lockcmds.c b/src/backend/commands/lockcmds.c
index f66b8f17b9b..416fb4fe7fe 100644
--- a/src/backend/commands/lockcmds.c
+++ b/src/backend/commands/lockcmds.c
@@ -72,7 +72,7 @@ RangeVarCallbackForLockTable(const RangeVar *rv, Oid relid, Oid oldrelid,
void *arg)
{
LOCKMODE lockmode = *(LOCKMODE *) arg;
- char relkind;
+ Relkind relkind;
char relpersistence;
AclResult aclresult;
@@ -190,7 +190,7 @@ LockViewRecurse_walker(Node *node, LockViewRecurse_context *context)
AclResult aclresult;
Oid relid = rte->relid;
- char relkind = rte->relkind;
+ Relkind relkind = rte->relkind;
char *relname = get_rel_name(relid);
/* Currently, we only allow plain tables or views to be locked. */
diff --git a/src/backend/commands/policy.c b/src/backend/commands/policy.c
index 21b8eebe32d..3aee2444a80 100644
--- a/src/backend/commands/policy.c
+++ b/src/backend/commands/policy.c
@@ -66,7 +66,7 @@ RangeVarCallbackForPolicy(const RangeVar *rv, Oid relid, Oid oldrelid,
{
HeapTuple tuple;
Form_pg_class classform;
- char relkind;
+ Relkind relkind;
tuple = SearchSysCache1(RELOID, ObjectIdGetDatum(relid));
if (!HeapTupleIsValid(tuple))
diff --git a/src/backend/commands/publicationcmds.c b/src/backend/commands/publicationcmds.c
index fc3a4c19e65..d8bb521f62e 100644
--- a/src/backend/commands/publicationcmds.c
+++ b/src/backend/commands/publicationcmds.c
@@ -1057,7 +1057,7 @@ AlterPublicationOptions(ParseState *pstate, AlterPublicationStmt *stmt,
{
Oid relid = lfirst_oid(lc);
HeapTuple rftuple;
- char relkind;
+ Relkind relkind;
char *relname;
bool has_rowfilter;
bool has_collist;
diff --git a/src/backend/commands/subscriptioncmds.c b/src/backend/commands/subscriptioncmds.c
index 33e3c25a50c..8403c4850a6 100644
--- a/src/backend/commands/subscriptioncmds.c
+++ b/src/backend/commands/subscriptioncmds.c
@@ -113,7 +113,7 @@ typedef struct SubOpts
typedef struct PublicationRelKind
{
RangeVar *rv;
- char relkind;
+ Relkind relkind;
} PublicationRelKind;
static List *fetch_relation_list(WalReceiverConn *wrconn, List *publications);
@@ -821,7 +821,7 @@ CreateSubscription(ParseState *pstate, CreateSubscriptionStmt *stmt,
foreach_ptr(PublicationRelKind, pubrelinfo, pubrels)
{
Oid relid;
- char relkind;
+ Relkind relkind;
RangeVar *rv = pubrelinfo->rv;
relid = RangeVarGetRelid(rv, AccessShareLock, false);
@@ -1014,7 +1014,7 @@ AlterSubscription_refresh(Subscription *sub, bool copy_data,
{
RangeVar *rv = pubrelinfo->rv;
Oid relid;
- char relkind;
+ Relkind relkind;
relid = RangeVarGetRelid(rv, AccessShareLock, false);
relkind = get_rel_relkind(relid);
@@ -2958,7 +2958,7 @@ fetch_relation_list(WalReceiverConn *wrconn, List *publications)
char *nspname;
char *relname;
bool isnull;
- char relkind;
+ Relkind relkind;
PublicationRelKind *relinfo = palloc_object(PublicationRelKind);
nspname = TextDatumGetCString(slot_getattr(slot, 1, &isnull));
diff --git a/src/backend/commands/tablecmds.c b/src/backend/commands/tablecmds.c
index f976c0e5c7e..7cdc6fa2e2a 100644
--- a/src/backend/commands/tablecmds.c
+++ b/src/backend/commands/tablecmds.c
@@ -170,7 +170,7 @@ typedef struct AlteredTableInfo
{
/* Information saved before any work commences: */
Oid relid; /* Relation to work on */
- char relkind; /* Its relkind */
+ Relkind relkind; /* Its relkind */
TupleDesc oldDesc; /* Pre-modification tuple descriptor */
/*
@@ -766,7 +766,7 @@ static void ATExecSplitPartition(List **wqueue, AlteredTableInfo *tab,
* ----------------------------------------------------------------
*/
ObjectAddress
-DefineRelation(CreateStmt *stmt, char relkind, Oid ownerId,
+DefineRelation(CreateStmt *stmt, Relkind relkind, Oid ownerId,
ObjectAddress *typaddress, const char *queryString)
{
char relname[NAMEDATALEN];
@@ -1534,7 +1534,7 @@ void
RemoveRelations(DropStmt *drop)
{
ObjectAddresses *objects;
- char relkind;
+ Relkind relkind;
ListCell *cell;
int flags = 0;
LOCKMODE lockmode = AccessExclusiveLock;
@@ -3792,7 +3792,7 @@ SetRelationTableSpace(Relation rel,
static void
renameatt_check(Oid myrelid, Form_pg_class classform, bool recursing)
{
- char relkind = classform->relkind;
+ Relkind relkind = classform->relkind;
if (classform->reloftype && !recursing)
ereport(ERROR,
@@ -4219,7 +4219,7 @@ RenameRelation(RenameStmt *stmt)
for (;;)
{
LOCKMODE lockmode;
- char relkind;
+ Relkind relkind;
bool obj_is_index;
lockmode = is_index_stmt ? ShareUpdateExclusiveLock : AccessExclusiveLock;
@@ -7257,7 +7257,7 @@ ATExecAddColumn(List **wqueue, AlteredTableInfo *tab, Relation rel,
Form_pg_class relform;
Form_pg_attribute attribute;
int newattnum;
- char relkind;
+ Relkind relkind;
Expr *defval;
List *children;
ListCell *child;
@@ -19522,7 +19522,7 @@ void
RangeVarCallbackMaintainsTable(const RangeVar *relation,
Oid relId, Oid oldRelId, void *arg)
{
- char relkind;
+ Relkind relkind;
AclResult aclresult;
/* Nothing to do if the relation was not found. */
@@ -19619,7 +19619,7 @@ RangeVarCallbackForAlterRelation(const RangeVar *rv, Oid relid, Oid oldrelid,
HeapTuple tuple;
Form_pg_class classform;
AclResult aclresult;
- char relkind;
+ Relkind relkind;
tuple = SearchSysCache1(RELOID, ObjectIdGetDatum(relid));
if (!HeapTupleIsValid(tuple))
diff --git a/src/backend/commands/trigger.c b/src/backend/commands/trigger.c
index 8df915f63fb..7bf826aa4aa 100644
--- a/src/backend/commands/trigger.c
+++ b/src/backend/commands/trigger.c
@@ -6178,7 +6178,7 @@ AfterTriggerSaveEvent(EState *estate, ResultRelInfo *relinfo,
TriggerDesc *trigdesc = relinfo->ri_TrigDesc;
AfterTriggerEventData new_event;
AfterTriggerSharedData new_shared;
- char relkind = rel->rd_rel->relkind;
+ Relkind relkind = rel->rd_rel->relkind;
int tgtype_event;
int tgtype_level;
int i;
diff --git a/src/backend/executor/nodeModifyTable.c b/src/backend/executor/nodeModifyTable.c
index f5e9d369940..4981aeba196 100644
--- a/src/backend/executor/nodeModifyTable.c
+++ b/src/backend/executor/nodeModifyTable.c
@@ -4351,7 +4351,7 @@ ExecModifyTable(PlanState *pstate)
if (operation == CMD_UPDATE || operation == CMD_DELETE ||
operation == CMD_MERGE)
{
- char relkind;
+ Relkind relkind;
Datum datum;
bool isNull;
@@ -4869,7 +4869,7 @@ ExecInitModifyTable(ModifyTable *node, EState *estate, int eflags)
if (operation == CMD_UPDATE || operation == CMD_DELETE ||
operation == CMD_MERGE)
{
- char relkind;
+ Relkind relkind;
relkind = resultRelInfo->ri_RelationDesc->rd_rel->relkind;
if (relkind == RELKIND_RELATION ||
diff --git a/src/backend/nodes/gen_node_support.pl b/src/backend/nodes/gen_node_support.pl
index 4308751f787..2e141fa3dcb 100644
--- a/src/backend/nodes/gen_node_support.pl
+++ b/src/backend/nodes/gen_node_support.pl
@@ -136,7 +136,8 @@ my @nodetag_only;
# types that are copied by straight assignment
my @scalar_types = qw(
bits32 bool char double int int8 int16 int32 int64 long uint8 uint16 uint32 uint64
- AclMode AttrNumber Cardinality Cost Index Oid RelFileNumber Selectivity Size StrategyNumber SubTransactionId TimeLineID XLogRecPtr
+ AclMode AttrNumber Cardinality Cost Index Oid RelFileNumber Relkind
+ Selectivity Size StrategyNumber SubTransactionId TimeLineID XLogRecPtr
);
# collect enum types
diff --git a/src/backend/optimizer/util/appendinfo.c b/src/backend/optimizer/util/appendinfo.c
index 689840d6564..c5396116738 100644
--- a/src/backend/optimizer/util/appendinfo.c
+++ b/src/backend/optimizer/util/appendinfo.c
@@ -956,7 +956,7 @@ add_row_identity_columns(PlannerInfo *root, Index rtindex,
Relation target_relation)
{
CmdType commandType = root->parse->commandType;
- char relkind = target_relation->rd_rel->relkind;
+ Relkind relkind = target_relation->rd_rel->relkind;
Var *var;
Assert(commandType == CMD_UPDATE || commandType == CMD_DELETE || commandType == CMD_MERGE);
diff --git a/src/backend/replication/pgoutput/pgoutput.c b/src/backend/replication/pgoutput/pgoutput.c
index e016f64e0b3..4f1152f7f48 100644
--- a/src/backend/replication/pgoutput/pgoutput.c
+++ b/src/backend/replication/pgoutput/pgoutput.c
@@ -2100,7 +2100,7 @@ get_rel_sync_entry(PGOutputData *data, Relation relation)
Oid publish_as_relid = relid;
int publish_ancestor_level = 0;
bool am_partition = get_rel_relispartition(relid);
- char relkind = get_rel_relkind(relid);
+ Relkind relkind = get_rel_relkind(relid);
List *rel_publications = NIL;
/* Reload publications if needed before use. */
diff --git a/src/backend/statistics/stat_utils.c b/src/backend/statistics/stat_utils.c
index 9c680f1cb37..9a9fcda6edd 100644
--- a/src/backend/statistics/stat_utils.c
+++ b/src/backend/statistics/stat_utils.c
@@ -148,7 +148,7 @@ RangeVarCallbackForStats(const RangeVar *relation,
Oid table_oid = relId;
HeapTuple tuple;
Form_pg_class form;
- char relkind;
+ Relkind relkind;
/*
* If we previously locked some other index's heap, and the name we're
diff --git a/src/backend/tcop/utility.c b/src/backend/tcop/utility.c
index 34dd6e18df5..240c663e11b 100644
--- a/src/backend/tcop/utility.c
+++ b/src/backend/tcop/utility.c
@@ -1501,7 +1501,7 @@ ProcessUtilitySlow(ParseState *pstate,
foreach(lc, inheritors)
{
Oid partrelid = lfirst_oid(lc);
- char relkind = get_rel_relkind(partrelid);
+ Relkind relkind = get_rel_relkind(partrelid);
if (relkind != RELKIND_RELATION &&
relkind != RELKIND_MATVIEW &&
diff --git a/src/backend/utils/activity/pgstat_relation.c b/src/backend/utils/activity/pgstat_relation.c
index bc8c43b96aa..33be4c05994 100644
--- a/src/backend/utils/activity/pgstat_relation.c
+++ b/src/backend/utils/activity/pgstat_relation.c
@@ -90,7 +90,7 @@ pgstat_copy_relation_stats(Relation dst, Relation src)
void
pgstat_init_relation(Relation rel)
{
- char relkind = rel->rd_rel->relkind;
+ Relkind relkind = rel->rd_rel->relkind;
/*
* We only count stats for relations with storage and partitioned tables
diff --git a/src/backend/utils/adt/acl.c b/src/backend/utils/adt/acl.c
index 3a6905f9546..e4e2f18fd62 100644
--- a/src/backend/utils/adt/acl.c
+++ b/src/backend/utils/adt/acl.c
@@ -2170,7 +2170,7 @@ has_sequence_privilege_name_id(PG_FUNCTION_ARGS)
Oid roleid;
AclMode mode;
AclResult aclresult;
- char relkind;
+ Relkind relkind;
bool is_missing = false;
roleid = get_role_oid_or_public(NameStr(*username));
@@ -2206,7 +2206,7 @@ has_sequence_privilege_id(PG_FUNCTION_ARGS)
Oid roleid;
AclMode mode;
AclResult aclresult;
- char relkind;
+ Relkind relkind;
bool is_missing = false;
roleid = GetUserId();
@@ -2269,7 +2269,7 @@ has_sequence_privilege_id_id(PG_FUNCTION_ARGS)
text *priv_type_text = PG_GETARG_TEXT_PP(2);
AclMode mode;
AclResult aclresult;
- char relkind;
+ Relkind relkind;
bool is_missing = false;
mode = convert_sequence_priv_string(priv_type_text);
diff --git a/src/backend/utils/adt/partitionfuncs.c b/src/backend/utils/adt/partitionfuncs.c
index e9db027aa2e..dff0ff2bb4c 100644
--- a/src/backend/utils/adt/partitionfuncs.c
+++ b/src/backend/utils/adt/partitionfuncs.c
@@ -33,7 +33,7 @@
static bool
check_rel_can_be_partition(Oid relid)
{
- char relkind;
+ Relkind relkind;
bool relispartition;
/* Check if relation exists */
@@ -109,7 +109,7 @@ pg_partition_tree(PG_FUNCTION_ARGS)
HeapTuple tuple;
Oid parentid = InvalidOid;
Oid relid = list_nth_oid(partitions, funcctx->call_cntr);
- char relkind = get_rel_relkind(relid);
+ Relkind relkind = get_rel_relkind(relid);
int level = 0;
List *ancestors = get_partition_ancestors(relid);
ListCell *lc;
diff --git a/src/backend/utils/adt/ruleutils.c b/src/backend/utils/adt/ruleutils.c
index b5a7ad9066e..ddd05ea2f86 100644
--- a/src/backend/utils/adt/ruleutils.c
+++ b/src/backend/utils/adt/ruleutils.c
@@ -1060,7 +1060,7 @@ pg_get_triggerdef_worker(Oid trigid, bool pretty)
if (!isnull)
{
Node *qual;
- char relkind;
+ Relkind relkind;
deparse_context context;
deparse_namespace dpns;
RangeTblEntry *oldrte;
diff --git a/src/backend/utils/cache/lsyscache.c b/src/backend/utils/cache/lsyscache.c
index b924a2d900b..09fe81880e7 100644
--- a/src/backend/utils/cache/lsyscache.c
+++ b/src/backend/utils/cache/lsyscache.c
@@ -2149,7 +2149,7 @@ get_rel_type_id(Oid relid)
*
* Returns the relkind associated with a given relation.
*/
-char
+Relkind
get_rel_relkind(Oid relid)
{
HeapTuple tp;
@@ -2158,9 +2158,9 @@ get_rel_relkind(Oid relid)
if (HeapTupleIsValid(tp))
{
Form_pg_class reltup = (Form_pg_class) GETSTRUCT(tp);
- char result;
+ Relkind result;
- result = reltup->relkind;
+ result = (Relkind) reltup->relkind;
ReleaseSysCache(tp);
return result;
}
diff --git a/src/backend/utils/cache/relcache.c b/src/backend/utils/cache/relcache.c
index 6b634c9fff1..e794f4b942f 100644
--- a/src/backend/utils/cache/relcache.c
+++ b/src/backend/utils/cache/relcache.c
@@ -3517,7 +3517,7 @@ RelationBuildLocalRelation(const char *relname,
bool shared_relation,
bool mapped_relation,
char relpersistence,
- char relkind)
+ Relkind relkind)
{
Relation rel;
MemoryContext oldcxt;
diff --git a/src/bin/pg_dump/pg_backup_archiver.c b/src/bin/pg_dump/pg_backup_archiver.c
index 35d3a07915d..12f917994f6 100644
--- a/src/bin/pg_dump/pg_backup_archiver.c
+++ b/src/bin/pg_dump/pg_backup_archiver.c
@@ -2687,7 +2687,7 @@ WriteToc(ArchiveHandle *AH)
WriteStr(AH, te->namespace);
WriteStr(AH, te->tablespace);
WriteStr(AH, te->tableam);
- WriteInt(AH, te->relkind);
+ WriteInt(AH, (char) te->relkind);
WriteStr(AH, te->owner);
WriteStr(AH, "false");
diff --git a/src/bin/pg_dump/pg_backup_archiver.h b/src/bin/pg_dump/pg_backup_archiver.h
index 325b53fc9bd..cb17aab4374 100644
--- a/src/bin/pg_dump/pg_backup_archiver.h
+++ b/src/bin/pg_dump/pg_backup_archiver.h
@@ -26,6 +26,7 @@
#include <time.h>
+#include "catalog/pg_class.h"
#include "libpq-fe.h"
#include "pg_backup.h"
#include "pqexpbuffer.h"
@@ -359,7 +360,7 @@ struct _tocEntry
char *tablespace; /* null if not in a tablespace; empty string
* means use database default */
char *tableam; /* table access method, only for TABLE tags */
- char relkind; /* relation kind, only for TABLE tags */
+ Relkind relkind; /* relation kind, only for TABLE tags */
char *owner;
char *desc;
char *defn;
@@ -404,7 +405,7 @@ typedef struct _archiveOpts
const char *namespace;
const char *tablespace;
const char *tableam;
- char relkind;
+ Relkind relkind;
const char *owner;
const char *description;
teSection section;
diff --git a/src/bin/pg_dump/pg_dump.c b/src/bin/pg_dump/pg_dump.c
index cdccccc1820..0c2cfb5abea 100644
--- a/src/bin/pg_dump/pg_dump.c
+++ b/src/bin/pg_dump/pg_dump.c
@@ -100,7 +100,7 @@ typedef struct
typedef struct
{
Oid oid; /* object OID */
- char relkind; /* object kind */
+ Relkind relkind; /* object kind */
RelFileNumber relfilenumber; /* object filenode */
Oid toast_oid; /* toast table OID */
RelFileNumber toast_relfilenumber; /* toast table filenode */
@@ -356,7 +356,7 @@ static void addBoundaryDependencies(DumpableObject **dobjs, int numObjs,
static void addConstrChildIdxDeps(DumpableObject *dobj, const IndxInfo *refidx);
static void getDomainConstraints(Archive *fout, TypeInfo *tyinfo);
-static void getTableData(DumpOptions *dopt, TableInfo *tblinfo, int numTables, char relkind);
+static void getTableData(DumpOptions *dopt, TableInfo *tblinfo, int numTables, Relkind relkind);
static void makeTableDataInfo(DumpOptions *dopt, TableInfo *tbinfo);
static void buildMatViewRefreshDependencies(Archive *fout);
static void getTableDataFKConstraints(void);
@@ -3031,7 +3031,7 @@ refreshMatViewData(Archive *fout, const TableDataInfo *tdinfo)
* set up dumpable objects representing the contents of tables
*/
static void
-getTableData(DumpOptions *dopt, TableInfo *tblinfo, int numTables, char relkind)
+getTableData(DumpOptions *dopt, TableInfo *tblinfo, int numTables, Relkind relkind)
{
int i;
@@ -5856,7 +5856,7 @@ collectBinaryUpgradeClassOids(Archive *fout)
for (int i = 0; i < nbinaryUpgradeClassOids; i++)
{
binaryUpgradeClassOids[i].oid = atooid(PQgetvalue(res, i, 0));
- binaryUpgradeClassOids[i].relkind = *PQgetvalue(res, i, 1);
+ binaryUpgradeClassOids[i].relkind = (Relkind) *PQgetvalue(res, i, 1);
binaryUpgradeClassOids[i].relfilenumber = atooid(PQgetvalue(res, i, 2));
binaryUpgradeClassOids[i].toast_oid = atooid(PQgetvalue(res, i, 3));
binaryUpgradeClassOids[i].toast_relfilenumber = atooid(PQgetvalue(res, i, 4));
@@ -7121,7 +7121,7 @@ getFuncs(Archive *fout)
static RelStatsInfo *
getRelationStatistics(Archive *fout, DumpableObject *rel, int32 relpages,
char *reltuples, int32 relallvisible,
- int32 relallfrozen, char relkind,
+ int32 relallfrozen, Relkind relkind,
char **indAttNames, int nindAttNames)
{
if (!fout->dopt->dumpStatistics)
diff --git a/src/bin/pg_dump/pg_dump.h b/src/bin/pg_dump/pg_dump.h
index 4c4b14e5fc7..99283983fac 100644
--- a/src/bin/pg_dump/pg_dump.h
+++ b/src/bin/pg_dump/pg_dump.h
@@ -215,7 +215,7 @@ typedef struct _typeInfo
Oid typelem;
Oid typrelid;
Oid typarray;
- char typrelkind; /* 'r', 'v', 'c', etc */
+ Relkind typrelkind; /* 'r', 'v', 'c', etc */
char typtype; /* 'b', 'c', etc */
bool isArray; /* true if auto-generated array type */
bool isMultirange; /* true if auto-generated multirange type */
@@ -307,7 +307,7 @@ typedef struct _tableInfo
DumpableObject dobj;
DumpableAcl dacl;
const char *rolname;
- char relkind;
+ Relkind relkind;
char relpersistence; /* relation persistence */
bool relispopulated; /* relation is populated */
char relreplident; /* replica identifier */
@@ -452,7 +452,7 @@ typedef struct _relStatsInfo
char *reltuples;
int32 relallvisible;
int32 relallfrozen;
- char relkind; /* 'r', 'm', 'i', etc */
+ Relkind relkind; /* 'r', 'm', 'i', etc */
/*
* indAttNames/nindAttNames are populated only if the relation is an index
diff --git a/src/bin/pgbench/pgbench.c b/src/bin/pgbench/pgbench.c
index 58735871c17..5920da28014 100644
--- a/src/bin/pgbench/pgbench.c
+++ b/src/bin/pgbench/pgbench.c
@@ -862,7 +862,7 @@ get_table_relkind(PGconn *con, const char *table)
{
PGresult *res;
char *val;
- char relkind;
+ Relkind relkind;
const char *params[1] = {table};
const char *sql =
"SELECT relkind FROM pg_catalog.pg_class WHERE oid=$1::pg_catalog.regclass";
diff --git a/src/bin/psql/describe.c b/src/bin/psql/describe.c
index f02a8df8875..9d1322ae891 100644
--- a/src/bin/psql/describe.c
+++ b/src/bin/psql/describe.c
@@ -41,7 +41,7 @@ static bool describeOneTableDetails(const char *schemaname,
const char *relationname,
const char *oid,
bool verbose);
-static void add_tablespace_footer(printTableContent *const cont, char relkind,
+static void add_tablespace_footer(printTableContent *const cont, Relkind relkind,
Oid tablespace, const bool newline);
static void add_role_attribute(PQExpBuffer buf, const char *const str);
static bool listTSParsersVerbose(const char *pattern);
@@ -1603,7 +1603,7 @@ describeOneTableDetails(const char *schemaname,
struct
{
int16 checks;
- char relkind;
+ Relkind relkind;
bool hasindex;
bool hasrules;
bool hastriggers;
@@ -3678,7 +3678,7 @@ error_return:
* footer.
*/
static void
-add_tablespace_footer(printTableContent *const cont, char relkind,
+add_tablespace_footer(printTableContent *const cont, Relkind relkind,
Oid tablespace, const bool newline)
{
/* relkinds for which we support tablespaces */
diff --git a/src/include/access/reloptions.h b/src/include/access/reloptions.h
index 0bd17b30ca7..eb7671f6ba5 100644
--- a/src/include/access/reloptions.h
+++ b/src/include/access/reloptions.h
@@ -22,6 +22,7 @@
#include "access/amapi.h"
#include "access/htup.h"
#include "access/tupdesc.h"
+#include "catalog/pg_class.h"
#include "nodes/pg_list.h"
#include "storage/lock.h"
@@ -248,7 +249,7 @@ extern void *build_local_reloptions(local_relopts *relopts, Datum options,
extern bytea *default_reloptions(Datum reloptions, bool validate,
relopt_kind kind);
-extern bytea *heap_reloptions(char relkind, Datum reloptions, bool validate);
+extern bytea *heap_reloptions(Relkind relkind, Datum reloptions, bool validate);
extern bytea *view_reloptions(Datum reloptions, bool validate);
extern bytea *partitioned_table_reloptions(Datum reloptions, bool validate);
extern bytea *index_reloptions(amoptions_function amoptions, Datum reloptions,
diff --git a/src/include/catalog/heap.h b/src/include/catalog/heap.h
index 624c415dadb..41ad4a57391 100644
--- a/src/include/catalog/heap.h
+++ b/src/include/catalog/heap.h
@@ -16,6 +16,7 @@
#include "catalog/indexing.h"
#include "catalog/objectaddress.h"
+#include "catalog/pg_class.h"
#include "parser/parse_node.h"
@@ -55,7 +56,7 @@ extern Relation heap_create(const char *relname,
RelFileNumber relfilenumber,
Oid accessmtd,
TupleDesc tupDesc,
- char relkind,
+ Relkind relkind,
char relpersistence,
bool shared_relation,
bool mapped_relation,
@@ -74,7 +75,7 @@ extern Oid heap_create_with_catalog(const char *relname,
Oid accessmtd,
TupleDesc tupdesc,
List *cooked_constraints,
- char relkind,
+ Relkind relkind,
char relpersistence,
bool shared_relation,
bool mapped_relation,
@@ -144,7 +145,7 @@ extern const FormData_pg_attribute *SystemAttributeDefinition(AttrNumber attno);
extern const FormData_pg_attribute *SystemAttributeByName(const char *attname);
-extern void CheckAttributeNamesTypes(TupleDesc tupdesc, char relkind,
+extern void CheckAttributeNamesTypes(TupleDesc tupdesc, Relkind relkind,
int flags);
extern void CheckAttributeType(const char *attname,
diff --git a/src/include/catalog/objectaddress.h b/src/include/catalog/objectaddress.h
index e2fe9db1161..77d444ba8d6 100644
--- a/src/include/catalog/objectaddress.h
+++ b/src/include/catalog/objectaddress.h
@@ -88,6 +88,6 @@ extern char *getObjectIdentityParts(const ObjectAddress *object,
bool missing_ok);
extern struct ArrayType *strlist_to_textarray(List *list);
-extern ObjectType get_relkind_objtype(char relkind);
+extern ObjectType get_relkind_objtype(Relkind relkind);
#endif /* OBJECTADDRESS_H */
diff --git a/src/include/catalog/pg_class.h b/src/include/catalog/pg_class.h
index 4afff1e8a4e..cd0d034f3d6 100644
--- a/src/include/catalog/pg_class.h
+++ b/src/include/catalog/pg_class.h
@@ -164,16 +164,19 @@ MAKE_SYSCACHE(RELNAMENSP, pg_class_relname_nsp_index, 128);
#ifdef EXPOSE_TO_CLIENT_CODE
-#define RELKIND_RELATION 'r' /* ordinary table */
-#define RELKIND_INDEX 'i' /* secondary index */
-#define RELKIND_SEQUENCE 'S' /* sequence object */
-#define RELKIND_TOASTVALUE 't' /* for out-of-line values */
-#define RELKIND_VIEW 'v' /* view */
-#define RELKIND_MATVIEW 'm' /* materialized view */
-#define RELKIND_COMPOSITE_TYPE 'c' /* composite type */
-#define RELKIND_FOREIGN_TABLE 'f' /* foreign table */
-#define RELKIND_PARTITIONED_TABLE 'p' /* partitioned table */
-#define RELKIND_PARTITIONED_INDEX 'I' /* partitioned index */
+typedef enum
+{
+ RELKIND_RELATION = 'r', /* ordinary table */
+ RELKIND_INDEX = 'i', /* secondary index */
+ RELKIND_SEQUENCE = 'S', /* sequence object */
+ RELKIND_TOASTVALUE = 't', /* for out-of-line values */
+ RELKIND_VIEW = 'v', /* view */
+ RELKIND_MATVIEW = 'm', /* materialized view */
+ RELKIND_COMPOSITE_TYPE = 'c', /* composite type */
+ RELKIND_FOREIGN_TABLE = 'f', /* foreign table */
+ RELKIND_PARTITIONED_TABLE = 'p', /* partitioned table */
+ RELKIND_PARTITIONED_INDEX = 'I', /* partitioned index */
+} Relkind;
/* annoying defines for client-side C string construction */
#define RELKIND_RELATION_STR "'r'"
@@ -246,6 +249,6 @@ MAKE_SYSCACHE(RELNAMENSP, pg_class_relname_nsp_index, 128);
#endif /* EXPOSE_TO_CLIENT_CODE */
-extern int errdetail_relkind_not_supported(char relkind);
+extern int errdetail_relkind_not_supported(Relkind relkind);
#endif /* PG_CLASS_H */
diff --git a/src/include/catalog/pg_publication.h b/src/include/catalog/pg_publication.h
index 368becca899..dc6a6d3d379 100644
--- a/src/include/catalog/pg_publication.h
+++ b/src/include/catalog/pg_publication.h
@@ -170,7 +170,7 @@ typedef enum PublicationPartOpt
extern List *GetPublicationRelations(Oid pubid, PublicationPartOpt pub_partopt);
extern List *GetAllTablesPublications(void);
-extern List *GetAllPublicationRelations(char relkind, bool pubviaroot);
+extern List *GetAllPublicationRelations(Relkind relkind, bool pubviaroot);
extern List *GetPublicationSchemas(Oid pubid);
extern List *GetSchemaPublications(Oid schemaid);
extern List *GetSchemaPublicationRelations(Oid schemaid,
diff --git a/src/include/commands/tablecmds.h b/src/include/commands/tablecmds.h
index ed38b311ddf..e65c2b45008 100644
--- a/src/include/commands/tablecmds.h
+++ b/src/include/commands/tablecmds.h
@@ -25,7 +25,7 @@ typedef struct AlterTableUtilityContext AlterTableUtilityContext; /* avoid inclu
* tcop/utility.h here */
-extern ObjectAddress DefineRelation(CreateStmt *stmt, char relkind, Oid ownerId,
+extern ObjectAddress DefineRelation(CreateStmt *stmt, Relkind relkind, Oid ownerId,
ObjectAddress *typaddress, const char *queryString);
extern TupleDesc BuildDescForRelation(const List *columns);
diff --git a/src/include/nodes/parsenodes.h b/src/include/nodes/parsenodes.h
index 646d6ced763..0d69eff8c83 100644
--- a/src/include/nodes/parsenodes.h
+++ b/src/include/nodes/parsenodes.h
@@ -22,6 +22,7 @@
#ifndef PARSENODES_H
#define PARSENODES_H
+#include "catalog/pg_class.h"
#include "common/relpath.h"
#include "nodes/bitmapset.h"
#include "nodes/lockoptions.h"
@@ -1147,7 +1148,7 @@ typedef struct RangeTblEntry
/* inheritance requested? */
bool inh;
/* relation kind (see pg_class.relkind) */
- char relkind pg_node_attr(query_jumble_ignore);
+ Relkind relkind pg_node_attr(query_jumble_ignore);
/* lock level that query requires on the rel */
int rellockmode pg_node_attr(query_jumble_ignore);
/* index of RTEPermissionInfo entry, or 0 */
diff --git a/src/include/replication/logicalproto.h b/src/include/replication/logicalproto.h
index 058a955e20c..e5539ffd46e 100644
--- a/src/include/replication/logicalproto.h
+++ b/src/include/replication/logicalproto.h
@@ -111,7 +111,7 @@ typedef struct LogicalRepRelation
char **attnames; /* column names */
Oid *atttyps; /* column types */
char replident; /* replica identity */
- char relkind; /* remote relation kind */
+ Relkind relkind; /* remote relation kind */
Bitmapset *attkeys; /* Bitmap of key columns */
} LogicalRepRelation;
diff --git a/src/include/utils/lsyscache.h b/src/include/utils/lsyscache.h
index 5655aca4c14..d7e1f6132db 100644
--- a/src/include/utils/lsyscache.h
+++ b/src/include/utils/lsyscache.h
@@ -16,6 +16,7 @@
#include "access/attnum.h"
#include "access/cmptype.h"
#include "access/htup.h"
+#include "catalog/pg_class.h"
#include "nodes/pg_list.h"
/* avoid including subscripting.h here */
@@ -142,7 +143,7 @@ extern Oid get_relname_relid(const char *relname, Oid relnamespace);
extern char *get_rel_name(Oid relid);
extern Oid get_rel_namespace(Oid relid);
extern Oid get_rel_type_id(Oid relid);
-extern char get_rel_relkind(Oid relid);
+extern Relkind get_rel_relkind(Oid relid);
extern bool get_rel_relispartition(Oid relid);
extern Oid get_rel_tablespace(Oid relid);
extern char get_rel_persistence(Oid relid);
diff --git a/src/include/utils/relcache.h b/src/include/utils/relcache.h
index 2700224939a..ced2c62580a 100644
--- a/src/include/utils/relcache.h
+++ b/src/include/utils/relcache.h
@@ -15,6 +15,7 @@
#define RELCACHE_H
#include "access/tupdesc.h"
+#include "catalog/pg_class.h"
#include "common/relpath.h"
#include "nodes/bitmapset.h"
@@ -120,7 +121,7 @@ extern Relation RelationBuildLocalRelation(const char *relname,
bool shared_relation,
bool mapped_relation,
char relpersistence,
- char relkind);
+ Relkind relkind);
/*
* Routines to manage assignment of new relfilenumber to a relation
--
2.47.3
--a5gwxv7edi5gnyjn--
^ permalink raw reply [nested|flat] 35+ messages in thread
* pg_rewind does not rewind diverging timelines
@ 2026-04-30 08:19 Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 3 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-04-30 08:19 UTC (permalink / raw)
To: pgsql-hackers@lists.postgresql.org
Hi all,
I have been playing around with various promotion scenarios to check if it
is possible to lose writes in more complicated scenarios involving
promotions and uses of synchronous_standby_names and decided to create a
TLA+ model for streaming replication involving promotions and check those
with TLC. You can find the models at [1] if you're interested.
There is one scenario that I assume is known that TLC found, but does not
seem to be fixed. It is a relatively rare case, but since the fix is quite
easy, I thought I'd share it with you and get feedback.
The scenario can occur if you're unlucky and have more than one crash when
promoting standbys to be primaries, and goes like this:
You have three servers, S1, S2, and S3. S1 is primary and S2 and S3 are
standbys. All are on timeline (TLI) 1.
1. S1 crashes
2. S1 recovers and starts promotion. It writes XLOG_END_OF_RECOVERY (EOR)
for TLI 2 to the WAL.
3. S1 It manages to write some records W1 to the WAL.
4. Before the EOR is replicated to any standby, S1 crashes again. It is now
on TLI 2 and has some changes that are not elsewhere.
5. S2 is promoted. It writes an EOR for TLI 2 (since it is not aware of any
other timeline) to the WAL.
6. S2 writes some records W2 to WAL and now S1 has a record of TLI 2
version 1 (TLI 2.1) and S2 is on TLI 2.2.
7. S1 recovers and wants to join as a standby. You run pg_rewind to get rid
of the extra data, but since S2 is also on TLI 2, pg_rewind will happily
assume that both are on the same timeline.
8. S2 is now a standby but has that extra record for W2 both in the WAL and
in the database.
The fix (see attached draft) is quite simple: add a UUID to the EOR and to
the history file. When comparing timelines, don't only check the TLI, also
check the UUID. If not both match, go back further until you find a
timeline where both the TLI and the timeline UUID matches and do the usual
fandango to find the good LSN to rewind to.
[1]: https://github.com/mkindahl/tla-postgres
Attachments:
[application/octet-stream] 0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch.v1 (28.5K, ../../CAN305gBeJr8m7ZRW9mH0zakEFR4hDUPDo8fJRKJOHWMORG5_Bg@mail.gmail.com/3-0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch.v1)
download
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-01 16:06 Mats Kindahl <mats.kindahl@gmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
2 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-05-01 16:06 UTC (permalink / raw)
To: pgsql-hackers@lists.postgresql.org
On Thu, Apr 30, 2026 at 10:19 AM Mats Kindahl <mats.kindahl@gmail.com>
wrote:
> Hi all,
>
> I have been playing around with various promotion scenarios to check if it
> is possible to lose writes in more complicated scenarios involving
> promotions and uses of synchronous_standby_names and decided to create a
> TLA+ model for streaming replication involving promotions and check those
> with TLC. You can find the models at [1] if you're interested.
>
> There is one scenario that I assume is known that TLC found, but does not
> seem to be fixed. It is a relatively rare case, but since the fix is quite
> easy, I thought I'd share it with you and get feedback.
>
> The scenario can occur if you're unlucky and have more than one crash when
> promoting standbys to be primaries, and goes like this:
>
> You have three servers, S1, S2, and S3. S1 is primary and S2 and S3 are
> standbys. All are on timeline (TLI) 1.
>
> 1. S1 crashes
> 2. S1 recovers and starts promotion. It writes XLOG_END_OF_RECOVERY (EOR)
> for TLI 2 to the WAL.
> 3. S1 It manages to write some records W1 to the WAL.
> 4. Before the EOR is replicated to any standby, S1 crashes again. It is
> now on TLI 2 and has some changes that are not elsewhere.
> 5. S2 is promoted. It writes an EOR for TLI 2 (since it is not aware of
> any other timeline) to the WAL.
> 6. S2 writes some records W2 to WAL and now S1 has a record of TLI 2
> version 1 (TLI 2.1) and S2 is on TLI 2.2.
> 7. S1 recovers and wants to join as a standby. You run pg_rewind to get
> rid of the extra data, but since S2 is also on TLI 2, pg_rewind will
> happily assume that both are on the same timeline.
> 8. S2 is now a standby but has that extra record for W2 both in the WAL
> and in the database.
>
> The fix (see attached draft) is quite simple: add a UUID to the EOR and to
> the history file. When comparing timelines, don't only check the TLI, also
> check the UUID. If not both match, go back further until you find a
> timeline where both the TLI and the timeline UUID matches and do the usual
> fandango to find the good LSN to rewind to.
>
> [1]: https://github.com/mkindahl/tla-postgres
>
Here is an updated version of the patch. It seems like it is not necessary
to extend the XLOG_END_OF_RECOVERY record with the UUID, just the history
files. The scenario is still the same though, and can trigger diverging
servers, possibly silent. I have an additional test case using a divergence
going back three promotions.
--
Best wishes,
Mats Kindahl, Multigres Developer, Supabase
Attachments:
[text/x-patch] v2.0002-pg_rewind-test-rewind-across-UUID-mismatched-TLI.patch (6.6K, ../../CAN305gC0VE8zB=guccMj-7cJTW4oOAmTYCktaUKSzyOup=HHEw@mail.gmail.com/3-v2.0002-pg_rewind-test-rewind-across-UUID-mismatched-TLI.patch)
download | inline diff:
From c6516844b6a7f37c656681f63ec8416433d56f0f Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Thu, 30 Apr 2026 07:06:34 +0200
Subject: pg_rewind: test rewind across UUID-mismatched TLI
Add a test for the case where the target has gone through three timelines
(TLI1 -> TLI2 -> TLI3) while the source independently promoted from
TLI1 to a numerically identical but UUID-distinct TLI2 (call it TLI2').
Without UUID detection, findCommonAncestorTimeline accepts TLI2 as the
common ancestor and begins its WAL scan from the TLI2 shutdown
checkpoint. That scan misses the 'x' INSERT that happened earlier in
TLI2, so the source page is never copied and 'x' survives the rewind.
With the UUID fix in the previous commit the algorithm detects the TLI2
/ TLI2' mismatch, backs up to TLI1 as the true common ancestor, and
starts the WAL scan from the last TLI1 checkpoint. The scan covers the
'x' INSERT, the source page (containing 'b') is copied, and the rewound
cluster ends up with only 'b' and 'origin' as expected.
The target has deliberately no insert on TLI3 to ensure the test
actually exercises the UUID-based ancestor search rather than passing by
coincidence (as it would if a TLI3 insert put the page into the scan
range even on unpatched code).
---
src/bin/pg_rewind/t/005_same_timeline.pl | 121 +++++++++++++++++++++++
1 file changed, 121 insertions(+)
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 539d05f57a1..7d49f60ebba 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -108,4 +108,125 @@ $node_a->teardown_node;
$node_b->teardown_node;
$node_origin->teardown_node;
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = PostgreSQL::Test::Cluster->new('origin2');
+$node_origin2->init(allows_streaming => 1);
+$node_origin2->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+$node_origin2->start;
+
+$node_origin2->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+$node_origin2->safe_psql('postgres', "INSERT INTO tbl VALUES ('origin')");
+$node_origin2->safe_psql('postgres', 'CHECKPOINT');
+
+# node_x and node_b both start from the same TLI 1 baseline.
+my $node_x = PostgreSQL::Test::Cluster->new('node_x');
+$node_origin2->backup('backup_x');
+$node_x->init_from_backup($node_origin2, 'backup_x', has_streaming => 1);
+$node_x->set_standby_mode();
+$node_x->start;
+
+my $node_b2 = PostgreSQL::Test::Cluster->new('node_b2');
+$node_origin2->backup('backup_b2');
+$node_b2->init_from_backup($node_origin2, 'backup_b2', has_streaming => 1);
+$node_b2->set_standby_mode();
+$node_b2->start;
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+$node_origin2->wait_for_catchup($node_x);
+$node_origin2->wait_for_catchup($node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+$node_x->safe_psql('postgres', "INSERT INTO tbl VALUES ('x')");
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my $node_a2 = PostgreSQL::Test::Cluster->new('node_a2');
+$node_x->backup('backup_a2');
+$node_a2->init_from_backup($node_x, 'backup_a2', has_streaming => 1);
+$node_a2->set_standby_mode();
+$node_a2->start;
+
+$node_x->wait_for_catchup($node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+$node_b2->safe_psql('postgres', "INSERT INTO tbl VALUES ('b')");
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+$node_a2->append_conf('postgresql.conf',
+ "restore_command = 'cp \"$node_x_waldir/%f\" \"%p\"'\n");
+
+$node_a2->stop;
+$node_b2->stop;
+
+my $node_a2_pgdata = $node_a2->data_dir;
+my $tmp_folder2 = PostgreSQL::Test::Utils::tempdir;
+copy("$node_a2_pgdata/postgresql.conf",
+ "$tmp_folder2/node_a2-postgresql.conf.tmp");
+
+command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $node_b2->data_dir,
+ '--target-pgdata' => $node_a2_pgdata,
+ '--no-sync',
+ '--restore-target-wal',
+ '--config-file' => "$tmp_folder2/node_a2-postgresql.conf.tmp",
+ ],
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1');
+
+move("$tmp_folder2/node_a2-postgresql.conf.tmp",
+ "$node_a2_pgdata/postgresql.conf");
+
+# node_a2 should now mirror node_b2: rows from TLI 2 and TLI 3 are gone,
+# replaced by node_b2's TLI 2' row.
+$node_a2->start;
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
done_testing();
--
2.43.0
[text/x-patch] v2.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (28.0K, ../../CAN305gC0VE8zB=guccMj-7cJTW4oOAmTYCktaUKSzyOup=HHEw@mail.gmail.com/4-v2.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 16456473c61537c5f8c7689a6dac340be6b84c43 Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Thu, 30 Apr 2026 07:05:36 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as "unknown / compatible", so
the new code is fully backward-compatible with old history files.
A new test in t/005_same_timeline.pl covers the same-TLI shortcut case:
two standbys independently promote to TLI 2, each with a distinct UUID.
---
src/backend/access/transam/timeline.c | 49 ++++++++-
src/backend/access/transam/xlog.c | 40 +++++++-
src/backend/utils/adt/uuid.c | 14 ++-
src/bin/pg_rewind/pg_rewind.c | 120 +++++++++++++++++++++--
src/bin/pg_rewind/t/005_same_timeline.pl | 87 ++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++++++++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 349 insertions(+), 24 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..bc768efa8a6 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -114,6 +116,7 @@ readTimeLineHistory(TimeLineID targetTLI)
entry = palloc_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
return list_make1(entry);
}
@@ -125,6 +128,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +159,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +169,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -182,6 +187,23 @@ readTimeLineHistory(TimeLineID targetTLI)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the
+ * reason string in field 4. It is in theory possible that the
+ * reason string starts with a UUID, but the current usage do
+ * not store a UUID. This allows us to support both old and new
+ * formats of history files without breaking compatibility by
+ * checking if the field contains a valid UUID.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+ pg_uuid_t *up = DatumGetUUIDP(datum);
+
+ memcpy(&entry->tluuid, up, sizeof(pg_uuid_t));
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -203,6 +225,7 @@ readTimeLineHistory(TimeLineID targetTLI)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
result = lcons(entry, result);
@@ -294,21 +317,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +433,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index e39af79c03b..586d996c56f 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -515,6 +516,13 @@ typedef struct XLogCtlData
TimeLineID InsertTimeLineID;
TimeLineID PrevTimeLineID;
+ /*
+ * UUID for the current promotion. Generated when the timeline history
+ * file is written and later embedded in the XLOG_END_OF_RECOVERY record.
+ * Protected by info_lck.
+ */
+ pg_uuid_t ThisTimeLineUUID;
+
/*
* SharedRecoveryState indicates if we're still in crash or archive
* recovery. Protected by info_lck.
@@ -6377,6 +6385,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ TimestampTz now = GetCurrentTimestamp();
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6407,8 +6418,27 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file and later into the
+ * XLOG_END_OF_RECOVERY record so that pg_rewind can distinguish two
+ * servers that independently promoted to the same timeline ID.
*/
+
+
+ /*
+ * TimestampTz is microseconds; generate_uuidv7 wants ms + sub-ms. We
+ * generate the UUID outside the spinlock, to avoid doing the relatively
+ * expensive UUID generation, which could involve unexpected delays,
+ * while holding the spinlock.
+ */
+ generate_uuidv7_r(&uuid_buf, (uint64) (now / 1000), (uint32) (now % 1000) * 1000);
+ SpinLockAcquire(&XLogCtl->info_lck);
+ memcpy(&XLogCtl->ThisTimeLineUUID, &uuid_buf, sizeof(pg_uuid_t));
+ SpinLockRelease(&XLogCtl->info_lck);
+
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
@@ -9042,8 +9072,16 @@ xlog_redo(XLogReaderState *record)
{
xl_end_of_recovery xlrec;
TimeLineID replayTLI;
+ uint32 rec_len;
- memcpy(&xlrec, XLogRecGetData(record), sizeof(xl_end_of_recovery));
+ /*
+ * Zero the struct first so that old records without UUID fields
+ * produce all-zero UUIDs, which pg_rewind treats as "unknown".
+ */
+ memset(&xlrec, 0, sizeof(xl_end_of_recovery));
+ rec_len = XLogRecGetDataLen(record);
+ memcpy(&xlrec, XLogRecGetData(record),
+ Min(rec_len, sizeof(xl_end_of_recovery)));
/*
* For Hot Standby, we could treat this like a Shutdown Checkpoint,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..8dc098d11e3 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,13 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +604,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..b34f62bf968 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by timelines_match().
+ */
+typedef struct TimelineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimelineHistoriesData;
+
+typedef TimelineHistoriesData * TimelineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimelineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimelineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,20 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to the
+ * same timeline ID: one crashed after writing the history file but before
+ * its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions carry
+ * different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point via
+ * findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +416,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -874,7 +903,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -920,6 +949,64 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). Zero on either side means "unknown; treat as matching" so
+ * that pg_rewind degrades gracefully when rewinding against an old server.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 || memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI greater than 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI; a zero UUID (old history file or
+ * synthetic entry) is treated as matching.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimelineHistories tlh)
+{
+ static const pg_uuid_t zero = {{0}};
+ pg_uuid_t *a,
+ *b;
+
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ a = &tlh->source[tlh->sourceNentries - 2].tluuid;
+ b = &tlh->target[tlh->targetNentries - 2].tluuid;
+
+ if (memcmp(a, &zero, UUID_LEN) == 0)
+ return true;
+ if (memcmp(b, &zero, UUID_LEN) == 0)
+ return true;
+
+ return memcmp(a, b, UUID_LEN) == 0;
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -936,17 +1023,30 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
/*
* Trace the history forward, until we hit the timeline diverge. It may
- * still be possible that the source and target nodes used the same
- * timeline number in their history but with different start position
- * depending on the history files that each node has fetched in previous
- * recovery processes. Hence check the start position of the new timeline
- * as well and move down by one extra timeline entry if they do not match.
+ * still be possible that the source and target nodes used the same timeline
+ * number in their history but with different start position depending on
+ * the history files that each node has fetched in previous recovery
+ * processes. Hence check the start position of the new timeline as well and
+ * move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history files
+ * with the same name (e.g. 00000003.history); in a shared WAL archive the
+ * second file silently overwrites the first. pg_rewind fetches each
+ * server's history file directly from that server, so it sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the promotion
+ * that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli. So to
+ * check whether entry[i] itself represents the same timeline on both sides
+ * we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is always the
+ * same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..539d05f57a1 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,89 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = PostgreSQL::Test::Cluster->new('origin');
+$node_origin->init(allows_streaming => 1);
+$node_origin->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+$node_origin->start;
+
+$node_origin->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+$node_origin->safe_psql('postgres', "INSERT INTO tbl VALUES ('initial')");
+$node_origin->safe_psql('postgres', 'CHECKPOINT');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my $node_a = PostgreSQL::Test::Cluster->new('node_a');
+$node_origin->backup('backup_a');
+$node_a->init_from_backup($node_origin, 'backup_a', has_streaming => 1);
+$node_a->set_standby_mode();
+$node_a->start;
+
+my $node_b = PostgreSQL::Test::Cluster->new('node_b');
+$node_origin->backup('backup_b');
+$node_b->init_from_backup($node_origin, 'backup_b', has_streaming => 1);
+$node_b->set_standby_mode();
+$node_b->start;
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+$node_origin->wait_for_catchup($node_a);
+$node_origin->wait_for_catchup($node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+$node_a->safe_psql('postgres', "INSERT INTO tbl VALUES ('in A')");
+$node_b->safe_psql('postgres', "INSERT INTO tbl VALUES ('in B')");
+
+# Stop both nodes; rewind node_a (target) from node_b (source) in local mode.
+$node_a->stop;
+$node_b->stop;
+
+my $node_a_pgdata = $node_a->data_dir;
+my $tmp_folder = PostgreSQL::Test::Utils::tempdir;
+copy("$node_a_pgdata/postgresql.conf",
+ "$tmp_folder/node_a-postgresql.conf.tmp");
+
+command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $node_b->data_dir,
+ '--target-pgdata' => $node_a_pgdata,
+ '--no-sync',
+ '--config-file' => "$tmp_folder/node_a-postgresql.conf.tmp",
+ ],
+ 'pg_rewind handles independent same-TLI promotion');
+
+move("$tmp_folder/node_a-postgresql.conf.tmp",
+ "$node_a_pgdata/postgresql.conf");
+
+# node_a should now mirror node_b: it has 'initial' and 'in B', not 'in A'.
+$node_a->start;
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\ninitial",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 13ae3ad4fbb..8d5e374dfad 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..784920c1f8e 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-21 22:09 surya poondla <suryapoondla4@gmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
2 siblings, 1 reply; 35+ messages in thread
From: surya poondla @ 2026-05-21 22:09 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org
Hi Mats,
Thanks for picking this up -- the scenario is a real one and I think the
UUID-tagging approach is a clean way to solve it. v2 applies and builds
without trouble, and the core algorithm reads well to me.
I have a handful of observations that I'd love your thoughts.
Regarding Correctness I have the below thoughts
1. UUIDv7 timestamp epoch.
In StartupXLOG():
TimestampTz now = GetCurrentTimestamp();
generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
(uint32)(now % 1000) * 1000);
I think there might be a small mismatch here: GetCurrentTimestamp() returns
microseconds since the Postgres epoch (2000-01-01),
whereas generate_uuidv7_r describes its first argument as milliseconds
since the Unix epoch.
As written that 30-year offset would land in the UUID's timestamp field, so
the resulting UUID wouldn't be a conformant UUIDv7 and wouldn't
time-order against UUIDv7s generated through the SQL functions.
Uniqueness is preserved either way, so the rewind logic still works as
intended but it seemed worth flagging.
I see conversion that's used elsewhere as:
us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
* SECS_PER_DAY * USECS_PER_SEC;
Or, since promotion isn't on a hot path, gettimeofday() / time(NULL)
directly would also be fine.
2. EOR-record path, the intent is unclear.
The comment above generate_uuidv7_r() at says:
"The same UUID is written into the history file and later into the
XLOG_END_OF_RECOVERY record so that pg_rewind can distinguish two
servers..."
But from what I can see only the history-file part actually lands.
xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't add
the UUID, and XLogCtl->ThisTimeLineUUID is written under info_lck without a
reader (I couldn't grep it).
The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads like
preparation for an EOR-struct extension that ended up not being part of the
patch.
Was the EOR-record piece something you intended to keep for a follow-up, or
has it been superseded by the history-file approach?
3. Malformed UUID handling in readTimeLineHistory().
The optional field-4 path is:
if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
{
Datum datum = DirectFunctionCall1(uuid_in,
CStringGetDatum(uuid_str));
...
}
uuid_in() raises ereport(ERROR) on a malformed input, while the surrounding
syntax-error paths in readTimeLineHistory() use FATAL deliberately.
In practice an ERROR during startup ends up being fatal too, so this isn't
strictly a bug but it would be nicer to stay consistent.
Regarding the Tests I have the following thoughts
The two new cases are nice, a few extensions that I think would strengthen
them:
1. A mixed-version case where one side has a zero UUID. That's the path
we're claiming is graceful, but nothing currently exercises it
2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that
findCommonAncestorTimeline's loop walks past matching entries
before hitting the mismatch. The 0002 test puts the divergence at
depth 1.
3. A small assertion against the on-disk 00000002.history contents, to pin
down the file format.
4. On 0002 the dependency on restore_command pointing at node_x's pg_wal is
the kind of thing that tends to break under
environment changes. A CHECKPOINT on node_x before the backup, or
wal_keep_size as in 0001, would let the test stand on its own.
I'm happy to keep reviewing/contributing, thanks again for working on it.
Regards,
Surya Poondla
>
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-24 18:30 Mats Kindahl <mats.kindahl@gmail.com>
parent: surya poondla <suryapoondla4@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-05-24 18:30 UTC (permalink / raw)
To: surya poondla <suryapoondla4@gmail.com>; +Cc: pgsql-hackers@lists.postgresql.org
On Fri, May 22, 2026 at 12:09 AM surya poondla <suryapoondla4@gmail.com>
wrote:
> Hi Mats,
>
> Thanks for picking this up -- the scenario is a real one and I think the
> UUID-tagging approach is a clean way to solve it. v2 applies and builds
> without trouble, and the core algorithm reads well to me.
> I have a handful of observations that I'd love your thoughts.
>
Hi Surya,
Thank you for the review. It is a quite rare scenario, but it is real and
the fix is simple.
> Regarding Correctness I have the below thoughts
>
> 1. UUIDv7 timestamp epoch.
> In StartupXLOG():
> TimestampTz now = GetCurrentTimestamp();
> generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
> (uint32)(now % 1000) * 1000);
>
> I think there might be a small mismatch here: GetCurrentTimestamp()
> returns microseconds since the Postgres epoch (2000-01-01),
> whereas generate_uuidv7_r describes its first argument as milliseconds
> since the Unix epoch.
> As written that 30-year offset would land in the UUID's timestamp field,
> so the resulting UUID wouldn't be a conformant UUIDv7 and wouldn't
> time-order against UUIDv7s generated through the SQL functions.
>
>
> Uniqueness is preserved either way, so the rewind logic still works as
> intended but it seemed worth flagging.
>
> I see conversion that's used elsewhere as:
> us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
> * SECS_PER_DAY * USECS_PER_SEC;
>
> Or, since promotion isn't on a hot path, gettimeofday() / time(NULL)
> directly would also be fine.
>
Yes, the intention was to use a proper timestamp to allow debugging servers
if necessary. Switched to gettimeofday() and used 0 for sub-ms since this
is not going to be critical. (We could use ns here as well, but that would
only solve a race if you have two servers being promoted in the same ms,
which I find unlikely, and there is a random number added for that
situation.)
> 2. EOR-record path, the intent is unclear.
>
> The comment above generate_uuidv7_r() at says:
>
> "The same UUID is written into the history file and later into the
> XLOG_END_OF_RECOVERY record so that pg_rewind can distinguish two
> servers..."
>
> But from what I can see only the history-file part actually lands.
> xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't add
> the UUID, and XLogCtl->ThisTimeLineUUID is written under info_lck without a
> reader (I couldn't grep it).
>
> The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads like
> preparation for an EOR-struct extension that ended up not being part of the
> patch.
>
> Was the EOR-record piece something you intended to keep for a follow-up,
> or has it been superseded by the history-file approach?
>
No, the EOR changes are not needed for the promotion, contrary to what I
originally thought. Cleaned up the comment and the code and removed all
traces of changes to the EOR (I hope).
>
>
> 3. Malformed UUID handling in readTimeLineHistory().
>
> The optional field-4 path is:
>
> if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> {
> Datum datum = DirectFunctionCall1(uuid_in,
> CStringGetDatum(uuid_str));
> ...
> }
>
> uuid_in() raises ereport(ERROR) on a malformed input, while the
> surrounding syntax-error paths in readTimeLineHistory() use FATAL
> deliberately.
> In practice an ERROR during startup ends up being fatal too, so this isn't
> strictly a bug but it would be nicer to stay consistent.
>
Agree. I added code to capture the error and raise a FATAL instead (with
the error message from the uuid_in, in case it is modified it makes sense
to show this).
> Regarding the Tests I have the following thoughts
>
> The two new cases are nice, a few extensions that I think would strengthen
> them:
> 1. A mixed-version case where one side has a zero UUID. That's the path
> we're claiming is graceful, but nothing currently exercises it
>
Yes, that should work regardless of whether the source or the target has
the zero UUID.
I realized one thing: if two timelines have identical TLI but one has zero
UUID and one has not, it seems they could not come from the same promotion
(one promotion happened on an old server and the other one on a new
server), that is, they should be treated as different. Does that make
sense? I made the necessary changes in the attached patches for testing.
Please have a look.
> 2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that
> findCommonAncestorTimeline's loop walks past matching entries
> before hitting the mismatch. The 0002 test puts the divergence at
> depth 1.
>
I was unsure if this test was necessary or interesting, hence a separate
commit. Since you thought it was useful, it's now rolled into the patch and
I extended the tests with the scenarios you suggested.
I also did some refactorings of the tests to avoid duplication. More below.
> 3. A small assertion against the on-disk 00000002.history contents, to pin
> down the file format.
> 4. On 0002 the dependency on restore_command pointing at node_x's pg_wal
> is the kind of thing that tends to break under
> environment changes. A CHECKPOINT on node_x before the backup, or
> wal_keep_size as in 0001, would let the test stand on its own.
>
Good point.
I refactored the code to avoid some duplication and make the test flow
self-explanatory and as part of that I set the wal_keep_size for all nodes.
In the process I noticed that many of the functions in RewindTest.pm do the
same job as the primitives I wrote, but have hard-coded variable names. I
could rewrite them to take parameters, but that would be quite a big patch
to add additional changes to each call site, so I did not do that and
rather added small wrappers specific for the tests in 005_same_timeline.pl.
Attached a new version of the now single patch.
> I'm happy to keep reviewing/contributing, thanks again for working on it.
>
Thank you for reviewing it.
--
Best wishes,
Mats Kindahl, Multigres Developer, Supabase
Attachments:
[text/x-patch] v3.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (37.7K, ../../CAN305gCcj4Mhr3uBQAnQCYsx6F-syp1rGtazoy=h+_EHO0xOXA@mail.gmail.com/3-v3.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 6077acc68ac944d0fee83fe20b6abd85ebaab686 Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Two new tests in t/005_same_timeline.pl cover both detection paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
---
src/backend/access/transam/timeline.c | 79 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 112 +++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 352 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 613 insertions(+), 23 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..df161dcc0d5 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,45 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +238,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +337,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +453,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index e39af79c03b..f0cf9f7b435 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6377,6 +6378,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6407,8 +6411,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..f1dc0196cd8 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +605,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..ddb1a6b5001 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by timelines_match().
+ */
+typedef struct TimelineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimelineHistoriesData;
+
+typedef TimelineHistoriesData * TimelineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimelineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimelineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,20 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +416,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -874,7 +903,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -920,6 +949,65 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimelineHistories tlh)
+{
+ static const pg_uuid_t zero = {0};
+ pg_uuid_t *a,
+ *b;
+
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ a = &tlh->source[tlh->sourceNentries - 2].tluuid;
+ b = &tlh->target[tlh->targetNentries - 2].tluuid;
+
+ if (memcmp(a, &zero, UUID_LEN) == 0 && memcmp(b, &zero, UUID_LEN) == 0)
+ return true;
+
+ return memcmp(a, b, UUID_LEN) == 0;
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -941,12 +1029,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..7d2dadfd171 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,354 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+sub rewind_node
+{
+ my ($target, $source, $label, @extra_args) = @_;
+ $source->stop;
+ $target->stop;
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start;
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+$node_a2->append_conf('postgresql.conf',
+ "restore_command = 'cp \"$node_x_waldir/%f\" \"%p\"'\n");
+
+rewind_node($node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ '--restore-target-wal');
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $side = $strip_target ? 'target' : 'source';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when target has zero UUID and $side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+$node_origin4->wait_for_catchup($node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 13ae3ad4fbb..8d5e374dfad 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..6839de2e0b2 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-25 05:20 Japin Li <japinli@hotmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Japin Li @ 2026-05-25 05:20 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi, Mats
On Sun, 24 May 2026 at 20:30, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> On Fri, May 22, 2026 at 12:09 AM surya poondla <suryapoondla4@gmail.com> wrote:
>
> Hi Mats,
>
> Thanks for picking this up -- the scenario is a real one and I think the UUID-tagging approach is a clean way to
> solve it. v2 applies and builds without trouble, and the core algorithm reads well to me.
> I have a handful of observations that I'd love your thoughts.
>
> Hi Surya,
>
> Thank you for the review. It is a quite rare scenario, but it is real and the fix is simple.
>
> Regarding Correctness I have the below thoughts
>
> 1. UUIDv7 timestamp epoch.
> In StartupXLOG():
> TimestampTz now = GetCurrentTimestamp();
> generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
> (uint32)(now % 1000) * 1000);
>
> I think there might be a small mismatch here: GetCurrentTimestamp() returns microseconds since the Postgres epoch
> (2000-01-01),
> whereas generate_uuidv7_r describes its first argument as milliseconds since the Unix epoch.
> As written that 30-year offset would land in the UUID's timestamp field, so the resulting UUID wouldn't be a
> conformant UUIDv7 and wouldn't
> time-order against UUIDv7s generated through the SQL functions.
>
>
>
> Uniqueness is preserved either way, so the rewind logic still works as intended but it seemed worth flagging.
>
> I see conversion that's used elsewhere as:
> us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
> * SECS_PER_DAY * USECS_PER_SEC;
>
> Or, since promotion isn't on a hot path, gettimeofday() / time(NULL) directly would also be fine.
>
> Yes, the intention was to use a proper timestamp to allow debugging servers if necessary. Switched to gettimeofday() and
> used 0 for sub-ms since this is not going to be critical. (We could use ns here as well, but that would only solve a race
> if you have two servers being promoted in the same ms, which I find unlikely, and there is a random number added for that
> situation.)
>
> 2. EOR-record path, the intent is unclear.
>
> The comment above generate_uuidv7_r() at says:
>
> "The same UUID is written into the history file and later into the XLOG_END_OF_RECOVERY record so that pg_rewind can
> distinguish two servers..."
>
> But from what I can see only the history-file part actually lands.
> xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't add the UUID, and XLogCtl->ThisTimeLineUUID is
> written under info_lck without a
> reader (I couldn't grep it).
>
> The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads like preparation for an EOR-struct extension that
> ended up not being part of the patch.
>
> Was the EOR-record piece something you intended to keep for a follow-up, or has it been superseded by the
> history-file approach?
>
> No, the EOR changes are not needed for the promotion, contrary to what I originally thought. Cleaned up the comment and
> the code and removed all traces of changes to the EOR (I hope).
>
>
>
> 3. Malformed UUID handling in readTimeLineHistory().
>
> The optional field-4 path is:
>
> if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> {
> Datum datum = DirectFunctionCall1(uuid_in,
> CStringGetDatum(uuid_str));
> ...
> }
>
> uuid_in() raises ereport(ERROR) on a malformed input, while the surrounding syntax-error paths in readTimeLineHistory
> () use FATAL deliberately.
> In practice an ERROR during startup ends up being fatal too, so this isn't strictly a bug but it would be nicer to
> stay consistent.
>
> Agree. I added code to capture the error and raise a FATAL instead (with the error message from the uuid_in, in case it
> is modified it makes sense to show this).
>
> Regarding the Tests I have the following thoughts
>
> The two new cases are nice, a few extensions that I think would strengthen them:
> 1. A mixed-version case where one side has a zero UUID. That's the path we're claiming is graceful, but nothing
> currently exercises it
>
> Yes, that should work regardless of whether the source or the target has the zero UUID.
>
> I realized one thing: if two timelines have identical TLI but one has zero UUID and one has not, it seems they could not
> come from the same promotion (one promotion happened on an old server and the other one on a new server), that is, they
> should be treated as different. Does that make sense? I made the necessary changes in the attached patches for testing.
> Please have a look.
>
> 2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that findCommonAncestorTimeline's loop walks past
> matching entries
> before hitting the mismatch. The 0002 test puts the divergence at depth 1.
>
> I was unsure if this test was necessary or interesting, hence a separate commit. Since you thought it was useful, it's
> now rolled into the patch and I extended the tests with the scenarios you suggested.
>
> I also did some refactorings of the tests to avoid duplication. More below.
>
> 3. A small assertion against the on-disk 00000002.history contents, to pin down the file format.
> 4. On 0002 the dependency on restore_command pointing at node_x's pg_wal is the kind of thing that tends to break
> under
> environment changes. A CHECKPOINT on node_x before the backup, or wal_keep_size as in 0001, would let the test
> stand on its own.
>
> Good point.
>
> I refactored the code to avoid some duplication and make the test flow self-explanatory and as part of that I set the
> wal_keep_size for all nodes.
>
> In the process I noticed that many of the functions in RewindTest.pm do the same job as the primitives I wrote, but have
> hard-coded variable names. I could rewrite them to take parameters, but that would be quite a big patch to add additional
> changes to each call site, so I did not do that and rather added small wrappers specific for the tests in
> 005_same_timeline.pl⚠️.
>
> Attached a new version of the now single patch.
>
> I'm happy to keep reviewing/contributing, thanks again for working on it.
>
> Thank you for reviewing it.
Thank you for your work. I have one comment.
+ a = &tlh->source[tlh->sourceNentries - 2].tluuid;
+ b = &tlh->target[tlh->targetNentries - 2].tluuid;
+
+ if (memcmp(a, &zero, UUID_LEN) == 0 && memcmp(b, &zero, UUID_LEN) == 0)
+ return true;
+
+ return memcmp(a, b, UUID_LEN) == 0;
Since we already have matchingTimelineUUID(), the above code can be simplified
using it.
--
Regards,
Japin Li
ChengDu WenWu Information Technology Co., Ltd.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-25 18:59 Mats Kindahl <mats.kindahl@gmail.com>
parent: Japin Li <japinli@hotmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-05-25 18:59 UTC (permalink / raw)
To: Japin Li <japinli@hotmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi Japin,
On Mon, May 25, 2026 at 7:21 AM Japin Li <japinli@hotmail.com> wrote:
>
> Hi, Mats
>
> On Sun, 24 May 2026 at 20:30, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> > On Fri, May 22, 2026 at 12:09 AM surya poondla <suryapoondla4@gmail.com>
> wrote:
> >
> > Hi Mats,
> >
> > Thanks for picking this up -- the scenario is a real one and I think
> the UUID-tagging approach is a clean way to
> > solve it. v2 applies and builds without trouble, and the core algorithm
> reads well to me.
> > I have a handful of observations that I'd love your thoughts.
> >
> > Hi Surya,
> >
> > Thank you for the review. It is a quite rare scenario, but it is real
> and the fix is simple.
> >
> > Regarding Correctness I have the below thoughts
> >
> > 1. UUIDv7 timestamp epoch.
> > In StartupXLOG():
> > TimestampTz now = GetCurrentTimestamp();
> > generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
> > (uint32)(now % 1000) * 1000);
> >
> > I think there might be a small mismatch here: GetCurrentTimestamp()
> returns microseconds since the Postgres epoch
> > (2000-01-01),
> > whereas generate_uuidv7_r describes its first argument as milliseconds
> since the Unix epoch.
> > As written that 30-year offset would land in the UUID's timestamp
> field, so the resulting UUID wouldn't be a
> > conformant UUIDv7 and wouldn't
> > time-order against UUIDv7s generated through the SQL functions.
> >
> >
> >
> > Uniqueness is preserved either way, so the rewind logic still works as
> intended but it seemed worth flagging.
> >
> > I see conversion that's used elsewhere as:
> > us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
> > * SECS_PER_DAY * USECS_PER_SEC;
> >
> > Or, since promotion isn't on a hot path, gettimeofday() / time(NULL)
> directly would also be fine.
> >
> > Yes, the intention was to use a proper timestamp to allow debugging
> servers if necessary. Switched to gettimeofday() and
> > used 0 for sub-ms since this is not going to be critical. (We could use
> ns here as well, but that would only solve a race
> > if you have two servers being promoted in the same ms, which I find
> unlikely, and there is a random number added for that
> > situation.)
> >
> > 2. EOR-record path, the intent is unclear.
> >
> > The comment above generate_uuidv7_r() at says:
> >
> > "The same UUID is written into the history file and later into the
> XLOG_END_OF_RECOVERY record so that pg_rewind can
> > distinguish two servers..."
> >
> > But from what I can see only the history-file part actually lands.
> > xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't
> add the UUID, and XLogCtl->ThisTimeLineUUID is
> > written under info_lck without a
> > reader (I couldn't grep it).
> >
> > The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads like
> preparation for an EOR-struct extension that
> > ended up not being part of the patch.
> >
> > Was the EOR-record piece something you intended to keep for a
> follow-up, or has it been superseded by the
> > history-file approach?
> >
> > No, the EOR changes are not needed for the promotion, contrary to what I
> originally thought. Cleaned up the comment and
> > the code and removed all traces of changes to the EOR (I hope).
> >
> >
> >
> > 3. Malformed UUID handling in readTimeLineHistory().
> >
> > The optional field-4 path is:
> >
> > if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> > {
> > Datum datum = DirectFunctionCall1(uuid_in,
> >
> CStringGetDatum(uuid_str));
> > ...
> > }
> >
> > uuid_in() raises ereport(ERROR) on a malformed input, while the
> surrounding syntax-error paths in readTimeLineHistory
> > () use FATAL deliberately.
> > In practice an ERROR during startup ends up being fatal too, so this
> isn't strictly a bug but it would be nicer to
> > stay consistent.
> >
> > Agree. I added code to capture the error and raise a FATAL instead (with
> the error message from the uuid_in, in case it
> > is modified it makes sense to show this).
> >
> > Regarding the Tests I have the following thoughts
> >
> > The two new cases are nice, a few extensions that I think would
> strengthen them:
> > 1. A mixed-version case where one side has a zero UUID. That's the path
> we're claiming is graceful, but nothing
> > currently exercises it
> >
> > Yes, that should work regardless of whether the source or the target has
> the zero UUID.
> >
> > I realized one thing: if two timelines have identical TLI but one has
> zero UUID and one has not, it seems they could not
> > come from the same promotion (one promotion happened on an old server
> and the other one on a new server), that is, they
> > should be treated as different. Does that make sense? I made the
> necessary changes in the attached patches for testing.
> > Please have a look.
> >
> > 2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that
> findCommonAncestorTimeline's loop walks past
> > matching entries
> > before hitting the mismatch. The 0002 test puts the divergence at
> depth 1.
> >
> > I was unsure if this test was necessary or interesting, hence a separate
> commit. Since you thought it was useful, it's
> > now rolled into the patch and I extended the tests with the scenarios
> you suggested.
> >
> > I also did some refactorings of the tests to avoid duplication. More
> below.
> >
> > 3. A small assertion against the on-disk 00000002.history contents, to
> pin down the file format.
> > 4. On 0002 the dependency on restore_command pointing at node_x's
> pg_wal is the kind of thing that tends to break
> > under
> > environment changes. A CHECKPOINT on node_x before the backup, or
> wal_keep_size as in 0001, would let the test
> > stand on its own.
> >
> > Good point.
> >
> > I refactored the code to avoid some duplication and make the test flow
> self-explanatory and as part of that I set the
> > wal_keep_size for all nodes.
> >
> > In the process I noticed that many of the functions in RewindTest.pm do
> the same job as the primitives I wrote, but have
> > hard-coded variable names. I could rewrite them to take parameters, but
> that would be quite a big patch to add additional
> > changes to each call site, so I did not do that and rather added small
> wrappers specific for the tests in
> > 005_same_timeline.pl⚠️.
> >
> > Attached a new version of the now single patch.
> >
> > I'm happy to keep reviewing/contributing, thanks again for working on
> it.
> >
> > Thank you for reviewing it.
>
> Thank you for your work. I have one comment.
>
> + a = &tlh->source[tlh->sourceNentries - 2].tluuid;
> + b = &tlh->target[tlh->targetNentries - 2].tluuid;
> +
> + if (memcmp(a, &zero, UUID_LEN) == 0 && memcmp(b, &zero, UUID_LEN)
> == 0)
> + return true;
> +
> + return memcmp(a, b, UUID_LEN) == 0;
>
> Since we already have matchingTimelineUUID(), the above code can be
> simplified
> using it.
>
Thank you for the review. I switched to using the matchingTimelineUUID()
for this part of the code and made some other minor improvements as well.
--
Best wishes,
Mats Kindahl, Multigres Developer, Supabase
Attachments:
[text/x-patch] v4.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (37.6K, ../../CAN305gBPFE8KPgT5cdsbK8Xwxxii_+Hp4WVhCWsjFOYJ9j4xaw@mail.gmail.com/3-v4.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 8e27263fc7769e50dbf682b269d9df78dbf1c85e Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Two new tests in t/005_same_timeline.pl cover both detection paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
---
src/backend/access/transam/timeline.c | 79 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 104 ++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 353 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 606 insertions(+), 23 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..df161dcc0d5 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,45 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +238,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +337,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +453,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index beddcb552d6..87486ec6a1c 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6378,6 +6379,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6408,8 +6412,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..f1dc0196cd8 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +605,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..9587cfd5792 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by timelines_match().
+ */
+typedef struct TimelineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimelineHistoriesData;
+
+typedef TimelineHistoriesData * TimelineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimelineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimelineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +417,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -874,7 +904,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -920,6 +950,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimelineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -941,12 +1021,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..52963f29e00 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,355 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+sub rewind_node
+{
+ my ($target, $source, $label, @extra_args) = @_;
+ $source->stop;
+ $target->stop;
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start;
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+$node_a2->append_conf('postgresql.conf',
+ "restore_command = 'cp \"$node_x_waldir/%f\" \"%p\"'\n");
+
+rewind_node($node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ '--restore-target-wal');
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 55663e6f4af..20a2f345fd3 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..6839de2e0b2 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-26 06:56 Japin Li <japinli@hotmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Japin Li @ 2026-05-26 06:56 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi, Mats
Thanks for updating the patch.
On Mon, 25 May 2026 at 20:59, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> Hi Japin,
>
> On Mon, May 25, 2026 at 7:21 AM Japin Li <japinli@hotmail.com> wrote:
>
> Hi, Mats
>
> On Sun, 24 May 2026 at 20:30, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> > On Fri, May 22, 2026 at 12:09 AM surya poondla <suryapoondla4@gmail.com> wrote:
> >
> > Hi Mats,
> >
> > Thanks for picking this up -- the scenario is a real one and I think the UUID-tagging approach is a clean way to
> > solve it. v2 applies and builds without trouble, and the core algorithm reads well to me.
> > I have a handful of observations that I'd love your thoughts.
> >
> > Hi Surya,
> >
> > Thank you for the review. It is a quite rare scenario, but it is real and the fix is simple.
> >
> > Regarding Correctness I have the below thoughts
> >
> > 1. UUIDv7 timestamp epoch.
> > In StartupXLOG():
> > TimestampTz now = GetCurrentTimestamp();
> > generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
> > (uint32)(now % 1000) * 1000);
> >
> > I think there might be a small mismatch here: GetCurrentTimestamp() returns microseconds since the Postgres epoch
> > (2000-01-01),
> > whereas generate_uuidv7_r describes its first argument as milliseconds since the Unix epoch.
> > As written that 30-year offset would land in the UUID's timestamp field, so the resulting UUID wouldn't be a
> > conformant UUIDv7 and wouldn't
> > time-order against UUIDv7s generated through the SQL functions.
> >
> >
> >
> > Uniqueness is preserved either way, so the rewind logic still works as intended but it seemed worth flagging.
> >
> > I see conversion that's used elsewhere as:
> > us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
> > * SECS_PER_DAY * USECS_PER_SEC;
> >
> > Or, since promotion isn't on a hot path, gettimeofday() / time(NULL) directly would also be fine.
> >
> > Yes, the intention was to use a proper timestamp to allow debugging servers if necessary. Switched to gettimeofday
> () and
> > used 0 for sub-ms since this is not going to be critical. (We could use ns here as well, but that would only solve
> a race
> > if you have two servers being promoted in the same ms, which I find unlikely, and there is a random number added
> for that
> > situation.)
> >
> > 2. EOR-record path, the intent is unclear.
> >
> > The comment above generate_uuidv7_r() at says:
> >
> > "The same UUID is written into the history file and later into the XLOG_END_OF_RECOVERY record so that pg_rewind
> can
> > distinguish two servers..."
> >
> > But from what I can see only the history-file part actually lands.
> > xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't add the UUID, and XLogCtl->ThisTimeLineUUID
> is
> > written under info_lck without a
> > reader (I couldn't grep it).
> >
> > The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads like preparation for an EOR-struct extension
> that
> > ended up not being part of the patch.
> >
> > Was the EOR-record piece something you intended to keep for a follow-up, or has it been superseded by the
> > history-file approach?
> >
> > No, the EOR changes are not needed for the promotion, contrary to what I originally thought. Cleaned up the comment
> and
> > the code and removed all traces of changes to the EOR (I hope).
> >
> >
> >
> > 3. Malformed UUID handling in readTimeLineHistory().
> >
> > The optional field-4 path is:
> >
> > if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> > {
> > Datum datum = DirectFunctionCall1(uuid_in,
> > CStringGetDatum(uuid_str));
> > ...
> > }
> >
> > uuid_in() raises ereport(ERROR) on a malformed input, while the surrounding syntax-error paths in
> readTimeLineHistory
> > () use FATAL deliberately.
> > In practice an ERROR during startup ends up being fatal too, so this isn't strictly a bug but it would be nicer to
> > stay consistent.
> >
> > Agree. I added code to capture the error and raise a FATAL instead (with the error message from the uuid_in, in
> case it
> > is modified it makes sense to show this).
> >
> > Regarding the Tests I have the following thoughts
> >
> > The two new cases are nice, a few extensions that I think would strengthen them:
> > 1. A mixed-version case where one side has a zero UUID. That's the path we're claiming is graceful, but nothing
> > currently exercises it
> >
> > Yes, that should work regardless of whether the source or the target has the zero UUID.
> >
> > I realized one thing: if two timelines have identical TLI but one has zero UUID and one has not, it seems they
> could not
> > come from the same promotion (one promotion happened on an old server and the other one on a new server), that is,
> they
> > should be treated as different. Does that make sense? I made the necessary changes in the attached patches for
> testing.
> > Please have a look.
> >
> > 2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that findCommonAncestorTimeline's loop walks past
> > matching entries
> > before hitting the mismatch. The 0002 test puts the divergence at depth 1.
> >
> > I was unsure if this test was necessary or interesting, hence a separate commit. Since you thought it was useful,
> it's
> > now rolled into the patch and I extended the tests with the scenarios you suggested.
> >
> > I also did some refactorings of the tests to avoid duplication. More below.
> >
> > 3. A small assertion against the on-disk 00000002.history contents, to pin down the file format.
> > 4. On 0002 the dependency on restore_command pointing at node_x's pg_wal is the kind of thing that tends to break
> > under
> > environment changes. A CHECKPOINT on node_x before the backup, or wal_keep_size as in 0001, would let the
> test
> > stand on its own.
> >
> > Good point.
> >
> > I refactored the code to avoid some duplication and make the test flow self-explanatory and as part of that I set
> the
> > wal_keep_size for all nodes.
> >
> > In the process I noticed that many of the functions in RewindTest.pm do the same job as the primitives I wrote, but
> have
> > hard-coded variable names. I could rewrite them to take parameters, but that would be quite a big patch to add
> additional
> > changes to each call site, so I did not do that and rather added small wrappers specific for the tests in
> > 005_same_timeline.pl⚠️⚠️.
> >
> > Attached a new version of the now single patch.
> >
> > I'm happy to keep reviewing/contributing, thanks again for working on it.
> >
> > Thank you for reviewing it.
>
> Thank you for your work. I have one comment.
>
> + a = &tlh->source[tlh->sourceNentries - 2].tluuid;
> + b = &tlh->target[tlh->targetNentries - 2].tluuid;
> +
> + if (memcmp(a, &zero, UUID_LEN) == 0 && memcmp(b, &zero, UUID_LEN) == 0)
> + return true;
> +
> + return memcmp(a, b, UUID_LEN) == 0;
>
> Since we already have matchingTimelineUUID(), the above code can be simplified
> using it.
>
> Thank you for the review. I switched to using the matchingTimelineUUID() for this part of the code and made some other
> minor improvements as well.
Here are some comments on v4.
1.
+/*
+ * Timeline histories for both clusters, populated by timelines_match().
+ */
I don't see a timelines_match() function. Does this refer to
matchAndFetchTimelines()?
2.
+typedef struct TimelineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimelineHistoriesData;
I'd prefer to use TimeLineHistoriesData to stay consistent with
TimeLineHistoryEntry. Anyway I'm not instant on it.
3.
+typedef TimelineHistoriesData * TimelineHistories;
The space between * and TimelineHistories is unnecessary — see
StringInfoData and other typedefs.
4.
+# node_x and node_b both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
There appears to be a typo in the comment. The node_b should be node_b2.
Everything else looks good. Thank you again for updating the patch!
> --
> Best wishes,
> Mats Kindahl, Multigres Developer, Supabase
--
Regards,
Japin Li
ChengDu WenWu Information Technology Co., Ltd.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-26 16:03 Mats Kindahl <mats.kindahl@gmail.com>
parent: Japin Li <japinli@hotmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-05-26 16:03 UTC (permalink / raw)
To: Japin Li <japinli@hotmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi Japin,
Thanks for reviewing the patch.
On Tue, May 26, 2026 at 8:56 AM Japin Li <japinli@hotmail.com> wrote:
>
> Hi, Mats
>
> Thanks for updating the patch.
>
> On Mon, 25 May 2026 at 20:59, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> > Hi Japin,
> >
> > On Mon, May 25, 2026 at 7:21 AM Japin Li <japinli@hotmail.com> wrote:
> >
> > Hi, Mats
> >
> > On Sun, 24 May 2026 at 20:30, Mats Kindahl <mats.kindahl@gmail.com>
> wrote:
> > > On Fri, May 22, 2026 at 12:09 AM surya poondla <
> suryapoondla4@gmail.com> wrote:
> > >
> > > Hi Mats,
> > >
> > > Thanks for picking this up -- the scenario is a real one and I think
> the UUID-tagging approach is a clean way to
> > > solve it. v2 applies and builds without trouble, and the core
> algorithm reads well to me.
> > > I have a handful of observations that I'd love your thoughts.
> > >
> > > Hi Surya,
> > >
> > > Thank you for the review. It is a quite rare scenario, but it is real
> and the fix is simple.
> > >
> > > Regarding Correctness I have the below thoughts
> > >
> > > 1. UUIDv7 timestamp epoch.
> > > In StartupXLOG():
> > > TimestampTz now = GetCurrentTimestamp();
> > > generate_uuidv7_r(&uuid_buf, (uint64)(now / 1000),
> > > (uint32)(now % 1000) * 1000);
> > >
> > > I think there might be a small mismatch here: GetCurrentTimestamp()
> returns microseconds since the Postgres epoch
> > > (2000-01-01),
> > > whereas generate_uuidv7_r describes its first argument as
> milliseconds since the Unix epoch.
> > > As written that 30-year offset would land in the UUID's timestamp
> field, so the resulting UUID wouldn't be a
> > > conformant UUIDv7 and wouldn't
> > > time-order against UUIDv7s generated through the SQL functions.
> > >
> > >
> > >
> > > Uniqueness is preserved either way, so the rewind logic still works
> as intended but it seemed worth flagging.
> > >
> > > I see conversion that's used elsewhere as:
> > > us = ts + (POSTGRES_EPOCH_JDATE - UNIX_EPOCH_JDATE)
> > > * SECS_PER_DAY * USECS_PER_SEC;
> > >
> > > Or, since promotion isn't on a hot path, gettimeofday() / time(NULL)
> directly would also be fine.
> > >
> > > Yes, the intention was to use a proper timestamp to allow debugging
> servers if necessary. Switched to gettimeofday
> > () and
> > > used 0 for sub-ms since this is not going to be critical. (We could
> use ns here as well, but that would only solve
> > a race
> > > if you have two servers being promoted in the same ms, which I find
> unlikely, and there is a random number added
> > for that
> > > situation.)
> > >
> > > 2. EOR-record path, the intent is unclear.
> > >
> > > The comment above generate_uuidv7_r() at says:
> > >
> > > "The same UUID is written into the history file and later into the
> XLOG_END_OF_RECOVERY record so that pg_rewind
> > can
> > > distinguish two servers..."
> > >
> > > But from what I can see only the history-file part actually lands.
> > > xl_end_of_recovery is unchanged, CreateEndOfRecoveryRecord() doesn't
> add the UUID, and XLogCtl->ThisTimeLineUUID
> > is
> > > written under info_lck without a
> > > reader (I couldn't grep it).
> > >
> > > The xlog_redo() memset() + Min(rec_len, sizeof(...)) change reads
> like preparation for an EOR-struct extension
> > that
> > > ended up not being part of the patch.
> > >
> > > Was the EOR-record piece something you intended to keep for a
> follow-up, or has it been superseded by the
> > > history-file approach?
> > >
> > > No, the EOR changes are not needed for the promotion, contrary to
> what I originally thought. Cleaned up the comment
> > and
> > > the code and removed all traces of changes to the EOR (I hope).
> > >
> > >
> > >
> > > 3. Malformed UUID handling in readTimeLineHistory().
> > >
> > > The optional field-4 path is:
> > >
> > > if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> > > {
> > > Datum datum = DirectFunctionCall1(uuid_in,
> > >
> CStringGetDatum(uuid_str));
> > > ...
> > > }
> > >
> > > uuid_in() raises ereport(ERROR) on a malformed input, while the
> surrounding syntax-error paths in
> > readTimeLineHistory
> > > () use FATAL deliberately.
> > > In practice an ERROR during startup ends up being fatal too, so this
> isn't strictly a bug but it would be nicer to
> > > stay consistent.
> > >
> > > Agree. I added code to capture the error and raise a FATAL instead
> (with the error message from the uuid_in, in
> > case it
> > > is modified it makes sense to show this).
> > >
> > > Regarding the Tests I have the following thoughts
> > >
> > > The two new cases are nice, a few extensions that I think would
> strengthen them:
> > > 1. A mixed-version case where one side has a zero UUID. That's the
> path we're claiming is graceful, but nothing
> > > currently exercises it
> > >
> > > Yes, that should work regardless of whether the source or the target
> has the zero UUID.
> > >
> > > I realized one thing: if two timelines have identical TLI but one has
> zero UUID and one has not, it seems they
> > could not
> > > come from the same promotion (one promotion happened on an old server
> and the other one on a new server), that is,
> > they
> > > should be treated as different. Does that make sense? I made the
> necessary changes in the attached patches for
> > testing.
> > > Please have a look.
> > >
> > > 2. A deeper-divergence case (e.g. TLI1->2->3 vs TLI1->2->3') so that
> findCommonAncestorTimeline's loop walks past
> > > matching entries
> > > before hitting the mismatch. The 0002 test puts the divergence
> at depth 1.
> > >
> > > I was unsure if this test was necessary or interesting, hence a
> separate commit. Since you thought it was useful,
> > it's
> > > now rolled into the patch and I extended the tests with the scenarios
> you suggested.
> > >
> > > I also did some refactorings of the tests to avoid duplication. More
> below.
> > >
> > > 3. A small assertion against the on-disk 00000002.history contents,
> to pin down the file format.
> > > 4. On 0002 the dependency on restore_command pointing at node_x's
> pg_wal is the kind of thing that tends to break
> > > under
> > > environment changes. A CHECKPOINT on node_x before the backup,
> or wal_keep_size as in 0001, would let the
> > test
> > > stand on its own.
> > >
> > > Good point.
> > >
> > > I refactored the code to avoid some duplication and make the test
> flow self-explanatory and as part of that I set
> > the
> > > wal_keep_size for all nodes.
> > >
> > > In the process I noticed that many of the functions in RewindTest.pm
> do the same job as the primitives I wrote, but
> > have
> > > hard-coded variable names. I could rewrite them to take parameters,
> but that would be quite a big patch to add
> > additional
> > > changes to each call site, so I did not do that and rather added
> small wrappers specific for the tests in
> > > 005_same_timeline.pl⚠️⚠️.
> > >
> > > Attached a new version of the now single patch.
> > >
> > > I'm happy to keep reviewing/contributing, thanks again for working
> on it.
> > >
> > > Thank you for reviewing it.
> >
> > Thank you for your work. I have one comment.
> >
> > + a = &tlh->source[tlh->sourceNentries - 2].tluuid;
> > + b = &tlh->target[tlh->targetNentries - 2].tluuid;
> > +
> > + if (memcmp(a, &zero, UUID_LEN) == 0 && memcmp(b, &zero,
> UUID_LEN) == 0)
> > + return true;
> > +
> > + return memcmp(a, b, UUID_LEN) == 0;
> >
> > Since we already have matchingTimelineUUID(), the above code can be
> simplified
> > using it.
> >
> > Thank you for the review. I switched to using the matchingTimelineUUID()
> for this part of the code and made some other
> > minor improvements as well.
>
> Here are some comments on v4.
>
> 1.
> +/*
> + * Timeline histories for both clusters, populated by timelines_match().
> + */
>
> I don't see a timelines_match() function. Does this refer to
> matchAndFetchTimelines()?
>
Correct. Updated.
>
> 2.
> +typedef struct TimelineHistoriesData
> +{
> + TimeLineHistoryEntry *source,
> + *target;
> + int sourceNentries,
> + targetNentries;
> +} TimelineHistoriesData;
>
> I'd prefer to use TimeLineHistoriesData to stay consistent with
> TimeLineHistoryEntry. Anyway I'm not instant on it.
>
Makes sense to be consistent. Updated.
>
> 3.
> +typedef TimelineHistoriesData * TimelineHistories;
>
> The space between * and TimelineHistories is unnecessary — see
> StringInfoData and other typedefs.
>
My mistake. FIxed.
> 4.
> +# node_x and node_b both start from the same TLI 1 baseline.
> +my ($node_x, $node_b2) =
> + setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
>
> There appears to be a typo in the comment. The node_b should be node_b2.
>
Right. Fixed.
>
>
> Everything else looks good. Thank you again for updating the patch!
>
Thank you again for reviewing the patch. :)
Attached a new version of the patch with the changes you suggested.
--
Best wishes,
Mats Kindahl, Multigres Developer, Supabase
Attachments:
[text/x-patch] v5.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (37.6K, ../../CAN305gCaErXmG3fg48n50dWUC7=ETBBopuFL_cgyzutXUdp-5g@mail.gmail.com/3-v5.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 42826e22a21f8818d3602fa31bb8acc311a0929b Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Two new tests in t/005_same_timeline.pl cover both detection paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
---
src/backend/access/transam/timeline.c | 79 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 104 ++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 353 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 606 insertions(+), 23 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..df161dcc0d5 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,45 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +238,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +337,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +453,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index beddcb552d6..87486ec6a1c 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6378,6 +6379,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6408,8 +6412,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..f1dc0196cd8 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +605,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..de5b77117d6 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by matchAndFetchTimelines().
+ */
+typedef struct TimeLineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimeLineHistoriesData;
+
+typedef TimeLineHistoriesData *TimeLineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimeLineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimeLineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +417,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -874,7 +904,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -920,6 +950,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimeLineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -941,12 +1021,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..9b8e8204775 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,355 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+sub rewind_node
+{
+ my ($target, $source, $label, @extra_args) = @_;
+ $source->stop;
+ $target->stop;
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start;
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b2 both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+$node_a2->append_conf('postgresql.conf',
+ "restore_command = 'cp \"$node_x_waldir/%f\" \"%p\"'\n");
+
+rewind_node($node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ '--restore-target-wal');
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 55663e6f4af..20a2f345fd3 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..6839de2e0b2 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-29 02:01 Japin Li <japinli@hotmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Japin Li @ 2026-05-29 02:01 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi, Mats
On Tue, 26 May 2026 at 18:03, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>
> Attached a new version of the patch with the changes you suggested.
>
I found an error on the Windows platform [1].
[07:08:28.538] >>> MALLOC_PERTURB_=168 PG_REGRESS=C:\cirrus\build\src/test\regress\pg_regress.exe REGRESS_SHLIB=C:\cirrus\build\src/test\regress\regress.dll MSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1 top_builddir=C:\cirrus\build UBSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1 MESON_TEST_ITERATION=1 PATH=C:\cirrus\build\tmp_install\usr\local\pgsql\bin;C:\cirrus\build\src\bin\pg_rewind;C:/cirrus/build/src/bin/pg_rewind/test;C:\VS_2019\VC\Tools\MSVC\14.29.30133\bin\HostX64\x64;C:\VS_2019\MSBuild\Current\bin\Roslyn;C:\Program Files (x86)\Windows Kits\10\bin\10.0.22621.0\x64;C:\Program Files (x86)\Windows Kits\10\bin\x64;C:\VS_2019\\MSBuild\Current\Bin;C:\Windows\Microsoft.NET\Framework64\v4.0.30319;C:\VS_2019\Common7\IDE\;C:\VS_2019\Common7\Tools\;C:\VS_2019\VC\Auxiliary\Build;C:\zstd\zstd-v1.5.2-win64;C:\zlib;C:\lz4;C:\icu;C:\winflexbison;C:\strawberry\5.42.0.1\perl\bin;C:\python\Scripts\;C:\python\;C:\Windows Kits\10\Debuggers\x64;C:\Program Files\Git\usr\bin;C:\Windows\system32;C:\Windows;C:\Windows\System32\Wbem;C:\Windows\System32\WindowsPowerShell\v1.0\;C:\Windows\System32\OpenSSH\;C:\ProgramData\GooGet;C:\Program Files\Google\Compute Engine\metadata_scripts;C:\Program Files (x86)\Google\Cloud SDK\google-cloud-sdk\bin;C:\Program Files\PowerShell\7\;C:\Program Files\Google\Compute Engine\sysprep;C:\ProgramData\chocolatey\bin;C:\Program Files\Git\cmd;C:\Program Files\Git\mingw64\bin;C:\Program Files\Git\usr\bin;C:\Windows\system32\config\systemprofile\AppData\Local\Microsoft\WindowsApps INITDB_TEMPLATE=C:/cirrus/build/tmp_install/initdb-template ASAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1 share_contrib_dir=C:/cirrus/build/tmp_install//usr/local/pgsql/share/contrib C:\python\python3.EXE C:\cirrus\build\..\src/tools/testwrap --basedir C:\cirrus\build --srcdir C:\cirrus\src\bin\pg_rewind --pg-test-extra --testgroup pg_rewind --testname 005_same_timeline -- C:\strawberry\5.42.0.1\perl\bin\perl.EXE -I C:/cirrus/src/test/perl -I C:\cirrus\src\bin\pg_rewind C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl
[07:08:28.538] ------------------------------------- 8< -------------------------------------
[07:08:28.538] stderr:
[07:08:28.538] # Failed test 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1'
[07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 45.
[07:08:28.538] # ---------- command failed ----------
[07:08:28.538] # pg_rewind --debug --source-pgdata C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_b2_data/pgdata --target-pgdata C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_a2_data/pgdata --no-sync --config-file C:\cirrus\build\testrun\pg_rewind\005_same_timeline\data\tmp_test_ZCeZ/target-postgresql.conf.tmp --restore-target-wal
[07:08:28.538] # -------------- stderr --------------
[07:08:28.538] # pg_rewind: using for rewind "restore_command = 'cp "C:cirrusuild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/%f" "%p"'"
[07:08:28.538] # pg_rewind: Source timeline history:
[07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
[07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/00000000
[07:08:28.538] # pg_rewind: Target timeline history:
[07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
[07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/060000E0
[07:08:28.538] # pg_rewind: 3: 0/060000E0 - 0/00000000
[07:08:28.538] # pg_rewind: servers diverged at WAL location 0/040000E0 on timeline 1
[07:08:28.538] # cp: cannot stat 'C:cirrus'$'\b''uild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/000000020000000000000004': No such file or directory
[07:08:28.538] # pg_rewind: error: could not restore file "000000020000000000000004" from archive
[07:08:28.538] # pg_rewind: error: could not find previous WAL record at 0/040000E0
[07:08:28.538] # ------------------------------------
[07:08:28.538] # Failed test 'rewound node reflects source history, not target TLI 2/TLI 3 data'
[07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 260.
[07:08:28.538] # got: 'origin2
[07:08:28.538] # x'
[07:08:28.538] # expected: 'b
[07:08:28.538] # origin2'
[07:08:28.538] # Looks like you failed 2 tests of 11.
[07:08:28.538]
[07:08:28.538] (test program exited with status code 2)
[07:08:28.538] ------------------------------------------------------------------------------
[07:08:28.538]
[1] https://cirrus-ci.com/task/6228217159221248
--
Regards,
Japin Li
ChengDu WenWu Information Technology Co., Ltd.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-05-30 20:26 Mats Kindahl <mats.kindahl@gmail.com>
parent: Japin Li <japinli@hotmail.com>
0 siblings, 2 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-05-30 20:26 UTC (permalink / raw)
To: Japin Li <japinli@hotmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi Japin,
On 5/29/26 04:01, Japin Li wrote:
> Hi, Mats
>
> On Tue, 26 May 2026 at 18:03, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>> Attached a new version of the patch with the changes you suggested.
>>
> I found an error on the Windows platform [1].
>
> [07:08:28.538] >>> MALLOC_PERTURB_=168 PG_REGRESS=C:\cirrus\build\src/test\regress\pg_regress.exe REGRESS_SHLIB=C:\cirrus\build\src/test\regress\regress.dll MSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1 top_builddir=C:\cirrus\build UBSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1 MESON_TEST_ITERATION=1 PATH=C:\cirrus\build\tmp_install\usr\local\pgsql\bin;C:\cirrus\build\src\bin\pg_rewind;C:/cirrus/build/src/bin/pg_rewind/test;C:\VS_2019\VC\Tools\MSVC\14.29.30133\bin\HostX64\x64;C:\VS_2019\MSBuild\Current\bin\Roslyn;C:\Program Files (x86)\Windows Kits\10\bin\10.0.22621.0\x64;C:\Program Files (x86)\Windows Kits\10\bin\x64;C:\VS_2019\\MSBuild\Current\Bin;C:\Windows\Microsoft.NET\Framework64\v4.0.30319;C:\VS_2019\Common7\IDE\;C:\VS_2019\Common7\Tools\;C:\VS_2019\VC\Auxiliary\Build;C:\zstd\zstd-v1.5.2-win64;C:\zlib;C:\lz4;C:\icu;C:\winflexbison;C:\strawberry\5.42.0.1\perl\bin;C:\python\Scripts\;C:\python\;C:\Windows Kits\10\Debuggers\x64;C:\Program Files\Git\usr\bin;C:\Windows\system32;C:\Windows;C:\Windows\System32\Wbem;C:\Windows\System32\WindowsPowerShell\v1.0\;C:\Windows\System32\OpenSSH\;C:\ProgramData\GooGet;C:\Program Files\Google\Compute Engine\metadata_scripts;C:\Program Files (x86)\Google\Cloud SDK\google-cloud-sdk\bin;C:\Program Files\PowerShell\7\;C:\Program Files\Google\Compute Engine\sysprep;C:\ProgramData\chocolatey\bin;C:\Program Files\Git\cmd;C:\Program Files\Git\mingw64\bin;C:\Program Files\Git\usr\bin;C:\Windows\system32\config\systemprofile\AppData\Local\Microsoft\WindowsApps INITDB_TEMPLATE=C:/cirrus/build/tmp_install/initdb-template ASAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1 share_contrib_dir=C:/cirrus/build/tmp_install//usr/local/pgsql/share/contrib C:\python\python3.EXE C:\cirrus\build\..\src/tools/testwrap --basedir C:\cirrus\build --srcdir C:\cirrus\src\bin\pg_rewind --pg-test-extra --testgroup pg_rewind --testname 005_same_timeline -- C:\strawberry\5.42.0.1\perl\bin\perl.EXE -I C:/cirrus/src/test/perl -I C:\cirrus\src\bin\pg_rewind C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl
> [07:08:28.538] ------------------------------------- 8< -------------------------------------
> [07:08:28.538] stderr:
> [07:08:28.538] # Failed test 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1'
> [07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 45.
> [07:08:28.538] # ---------- command failed ----------
> [07:08:28.538] # pg_rewind --debug --source-pgdata C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_b2_data/pgdata --target-pgdata C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_a2_data/pgdata --no-sync --config-file C:\cirrus\build\testrun\pg_rewind\005_same_timeline\data\tmp_test_ZCeZ/target-postgresql.conf.tmp --restore-target-wal
> [07:08:28.538] # -------------- stderr --------------
> [07:08:28.538] # pg_rewind: using for rewind "restore_command = 'cp "C:cirrusuild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/%f" "%p"'"
> [07:08:28.538] # pg_rewind: Source timeline history:
> [07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
> [07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/00000000
> [07:08:28.538] # pg_rewind: Target timeline history:
> [07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
> [07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/060000E0
> [07:08:28.538] # pg_rewind: 3: 0/060000E0 - 0/00000000
> [07:08:28.538] # pg_rewind: servers diverged at WAL location 0/040000E0 on timeline 1
> [07:08:28.538] # cp: cannot stat 'C:cirrus'$'\b''uild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/000000020000000000000004': No such file or directory
> [07:08:28.538] # pg_rewind: error: could not restore file "000000020000000000000004" from archive
> [07:08:28.538] # pg_rewind: error: could not find previous WAL record at 0/040000E0
> [07:08:28.538] # ------------------------------------
> [07:08:28.538] # Failed test 'rewound node reflects source history, not target TLI 2/TLI 3 data'
> [07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 260.
> [07:08:28.538] # got: 'origin2
> [07:08:28.538] # x'
> [07:08:28.538] # expected: 'b
> [07:08:28.538] # origin2'
> [07:08:28.538] # Looks like you failed 2 tests of 11.
> [07:08:28.538]
> [07:08:28.538] (test program exited with status code 2)
> [07:08:28.538] ------------------------------------------------------------------------------
> [07:08:28.538]
>
>
> [1] https://cirrus-ci.com/task/6228217159221248
Thanks for testing it on Windows.
It seems like the path needs to be cleaned on Windows. I checked
Cluster.pm and created a version of that code and added that to the test
that should work. See attached patch.
I noted that many of the paths are not platform-agnostic. It an idea to
switch to use something like File::Spec instead and build paths using
that, but it's out of scope for this patch.
Best wishes,
Mats Kindahl, Multigres Engineer, Supabase
Attachments:
[text/x-patch] v6.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (37.8K, ../../9ce0d2b9-7a41-4a8a-b299-da295bb4514f@gmail.com/2-v6.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 5bc2a32054e13c579ef0045e08721332288aed64 Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Two new tests in t/005_same_timeline.pl cover both detection paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
---
src/backend/access/transam/timeline.c | 79 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 104 ++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 362 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 615 insertions(+), 23 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..df161dcc0d5 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,45 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +238,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +337,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +453,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index beddcb552d6..87486ec6a1c 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6378,6 +6379,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6408,8 +6412,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..f1dc0196cd8 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +605,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..de5b77117d6 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by matchAndFetchTimelines().
+ */
+typedef struct TimeLineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimeLineHistoriesData;
+
+typedef TimeLineHistoriesData *TimeLineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimeLineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimeLineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +417,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -874,7 +904,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -920,6 +950,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimeLineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -941,12 +1021,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..2360c3df1d0 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,364 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+sub rewind_node
+{
+ my ($target, $source, $label, @extra_args) = @_;
+ $source->stop;
+ $target->stop;
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start;
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b2 both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+if ($PostgreSQL::Test::Utils::windows_os)
+{
+ $node_x_waldir =~ s{\\}{\\\\}g;
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'copy "$node_x_waldir\\\\%f" "%p"'\n));
+}
+else
+{
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'cp "$node_x_waldir/%f" "%p"'\n));
+}
+
+rewind_node($node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ '--restore-target-wal');
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 55663e6f4af..20a2f345fd3 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..6839de2e0b2 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-01 02:59 Japin Li <japinli@hotmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
1 sibling, 1 reply; 35+ messages in thread
From: Japin Li @ 2026-06-01 02:59 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: surya poondla <suryapoondla4@gmail.com>; pgsql-hackers@lists.postgresql.org
Hi, Mats
On Sat, 30 May 2026 at 22:26, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> Hi Japin,
>
> On 5/29/26 04:01, Japin Li wrote:
>
>> Hi, Mats
>>
>> On Tue, 26 May 2026 at 18:03, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>>> Attached a new version of the patch with the changes you suggested.
>>>
>> I found an error on the Windows platform [1].
>>
>> [07:08:28.538] >>> MALLOC_PERTURB_=168
>> PG_REGRESS=C:\cirrus\build\src/test\regress\pg_regress.exe
>> REGRESS_SHLIB=C:\cirrus\build\src/test\regress\regress.dll
>> MSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1
>> top_builddir=C:\cirrus\build
>> UBSAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1:print_stacktrace=1
>> MESON_TEST_ITERATION=1
>> PATH=C:\cirrus\build\tmp_install\usr\local\pgsql\bin;C:\cirrus\build\src\bin\pg_rewind;C:/cirrus/build/src/bin/pg_rewind/test;C:\VS_2019\VC\Tools\MSVC\14.29.30133\bin\HostX64\x64;C:\VS_2019\MSBuild\Current\bin\Roslyn;C:\Program
>> Files (x86)\Windows Kits\10\bin\10.0.22621.0\x64;C:\Program Files
>> (x86)\Windows
>> Kits\10\bin\x64;C:\VS_2019\\MSBuild\Current\Bin;C:\Windows\Microsoft.NET\Framework64\v4.0.30319;C:\VS_2019\Common7\IDE\;C:\VS_2019\Common7\Tools\;C:\VS_2019\VC\Auxiliary\Build;C:\zstd\zstd-v1.5.2-win64;C:\zlib;C:\lz4;C:\icu;C:\winflexbison;C:\strawberry\5.42.0.1\perl\bin;C:\python\Scripts\;C:\python\;C:\Windows
>> Kits\10\Debuggers\x64;C:\Program
>> Files\Git\usr\bin;C:\Windows\system32;C:\Windows;C:\Windows\System32\Wbem;C:\Windows\System32\WindowsPowerShell\v1.0\;C:\Windows\System32\OpenSSH\;C:\ProgramData\GooGet;C:\Program
>> Files\Google\Compute Engine\metadata_scripts;C:\Program Files
>> (x86)\Google\Cloud SDK\google-cloud-sdk\bin;C:\Program
>> Files\PowerShell\7\;C:\Program Files\Google\Compute
>> Engine\sysprep;C:\ProgramData\chocolatey\bin;C:\Program
>> Files\Git\cmd;C:\Program Files\Git\mingw64\bin;C:\Program
>> Files\Git\usr\bin;C:\Windows\system32\config\systemprofile\AppData\Local\Microsoft\WindowsApps
>> INITDB_TEMPLATE=C:/cirrus/build/tmp_install/initdb-template
>> ASAN_OPTIONS=halt_on_error=1:abort_on_error=1:print_summary=1
>> share_contrib_dir=C:/cirrus/build/tmp_install//usr/local/pgsql/share/contrib
>> C:\python\python3.EXE C:\cirrus\build\..\src/tools/testwrap
>> --basedir C:\cirrus\build --srcdir C:\cirrus\src\bin\pg_rewind
>> --pg-test-extra --testgroup pg_rewind --testname 005_same_timeline
>> -- C:\strawberry\5.42.0.1\perl\bin\perl.EXE -I
>> C:/cirrus/src/test/perl -I C:\cirrus\src\bin\pg_rewind
>> C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl
>> [07:08:28.538] ------------------------------------- 8< -------------------------------------
>> [07:08:28.538] stderr:
>> [07:08:28.538] # Failed test 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1'
>> [07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 45.
>> [07:08:28.538] # ---------- command failed ----------
>> [07:08:28.538] # pg_rewind --debug --source-pgdata
>> C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_b2_data/pgdata
>> --target-pgdata
>> C:\cirrus\build/testrun/pg_rewind/005_same_timeline\data/t_005_same_timeline_node_a2_data/pgdata
>> --no-sync --config-file
>> C:\cirrus\build\testrun\pg_rewind\005_same_timeline\data\tmp_test_ZCeZ/target-postgresql.conf.tmp
>> --restore-target-wal
>> [07:08:28.538] # -------------- stderr --------------
>> [07:08:28.538] # pg_rewind: using for rewind "restore_command = 'cp "C:cirrusuild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/%f" "%p"'"
>> [07:08:28.538] # pg_rewind: Source timeline history:
>> [07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
>> [07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/00000000
>> [07:08:28.538] # pg_rewind: Target timeline history:
>> [07:08:28.538] # pg_rewind: 1: 0/00000000 - 0/040000E0
>> [07:08:28.538] # pg_rewind: 2: 0/040000E0 - 0/060000E0
>> [07:08:28.538] # pg_rewind: 3: 0/060000E0 - 0/00000000
>> [07:08:28.538] # pg_rewind: servers diverged at WAL location 0/040000E0 on timeline 1
>> [07:08:28.538] # cp: cannot stat 'C:cirrus'$'\b''uild/testrun/pg_rewind/005_same_timelinedata/t_005_same_timeline_node_x_data/pgdata/pg_wal/000000020000000000000004': No such file or directory
>> [07:08:28.538] # pg_rewind: error: could not restore file "000000020000000000000004" from archive
>> [07:08:28.538] # pg_rewind: error: could not find previous WAL record at 0/040000E0
>> [07:08:28.538] # ------------------------------------
>> [07:08:28.538] # Failed test 'rewound node reflects source history, not target TLI 2/TLI 3 data'
>> [07:08:28.538] # at C:/cirrus/src/bin/pg_rewind/t/005_same_timeline.pl line 260.
>> [07:08:28.538] # got: 'origin2
>> [07:08:28.538] # x'
>> [07:08:28.538] # expected: 'b
>> [07:08:28.538] # origin2'
>> [07:08:28.538] # Looks like you failed 2 tests of 11.
>> [07:08:28.538]
>> [07:08:28.538] (test program exited with status code 2)
>> [07:08:28.538] ------------------------------------------------------------------------------
>> [07:08:28.538]
>>
>>
>> [1] https://cirrus-ci.com/task/6228217159221248
>
> Thanks for testing it on Windows.
>
> It seems like the path needs to be cleaned on Windows. I checked
> Cluster.pm and created a version of that code and added that to the
> test that should work. See attached patch.
>
> I noted that many of the paths are not platform-agnostic. It an idea
> to switch to use something like File::Spec instead and build paths
> using that, but it's out of scope for this patch.
>
Thanks for updating the patch. LGTM.
--
Regards,
Japin Li
ChengDu WenWu Information Technology Co., Ltd.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-01 06:30 Kyotaro Horiguchi <horikyota.ntt@gmail.com>
parent: Japin Li <japinli@hotmail.com>
0 siblings, 4 replies; 35+ messages in thread
From: Kyotaro Horiguchi @ 2026-06-01 06:30 UTC (permalink / raw)
To: japinli@hotmail.com; +Cc: mats.kindahl@gmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
Sorry, I only just noticed this thread.
I may be missing something, but UUID feels somewhat heavyweight to me
for this problem.
I wonder whether strengthening the history-based matching would be
sufficient instead. If timelines with the same TLI but different
histories can be treated as distinct and pg_rewind continues walking
the history chain until it finds a common ancestor, that seems like a
fairly natural fit with the existing timeline model.
UUIDs would certainly make identification straightforward, although
they would also introduce longer identifiers that are a bit less
convenient for humans to work with. My initial thought is that it may
be worth exploring how far we can get with the existing history
information before introducing a new identifier.
regards.
--
Kyotaro Horiguchi
NTT Open Source Software Center
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-01 20:32 surya poondla <suryapoondla4@gmail.com>
parent: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
3 siblings, 0 replies; 35+ messages in thread
From: surya poondla @ 2026-06-01 20:32 UTC (permalink / raw)
To: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: japinli@hotmail.com; mats.kindahl@gmail.com; pgsql-hackers@lists.postgresql.org
Hi Mats, Japin
Nice points were discussed and addressed.
The latest patch looks good to me.
Regards,
Surya Poondla
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-02 02:13 Mats Kindahl <mats.kindahl@gmail.com>
parent: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
3 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-06-02 02:13 UTC (permalink / raw)
To: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; japinli@hotmail.com; +Cc: suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 6/1/26 08:30, Kyotaro Horiguchi wrote:
> Sorry, I only just noticed this thread.
>
> I may be missing something, but UUID feels somewhat heavyweight to me
> for this problem.
>
> I wonder whether strengthening the history-based matching would be
> sufficient instead. If timelines with the same TLI but different
> histories can be treated as distinct and pg_rewind continues walking
> the history chain until it finds a common ancestor, that seems like a
> fairly natural fit with the existing timeline model.
> UUIDs would certainly make identification straightforward, although
> they would also introduce longer identifiers that are a bit less
> convenient for humans to work with. My initial thought is that it may
> be worth exploring how far we can get with the existing history
> information before introducing a new identifier.
It is a good idea, but unfortunately there are positions in the timeline
that have same TLI, same LSN, but are still different timelines because
they originate from different promotions.
Just to summarize the situation: the timeline history file contains a
TLI (which is a number), and a switchpoint (which is an LSN). Each time
pg_promote is called, a new timeline is created based on the previous
TLI (it is increased by 1) and the LSN at that point. (The actual
history file is written by StartupXLOG, not by pg_promote, but
pg_promote triggers the process by writing a marker file.)
If two servers go through the same sequence, e.g., start at the same
timeline, does a promote, and write same length but different data
(e.g., add a line to a table, but with different contents), they might
end up with same TLI, same LSN, but different pg_promote calls, and
different database contents, hence it is not possible to distinguish them.
LSNs are usually different, so it is not a very likely scenario, but it
is still there.
The UUID is just generated and written when pg_promote is called, which
is not very often, hence does not affect the server and replication very
often. Note that the UUID is _not_ in the EOR (EndOfRecovery) record,
just in the timeline history file.
Best wishes,
Mats Kindahl
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-02 05:29 Kyotaro Horiguchi <horikyota.ntt@gmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 0 replies; 35+ messages in thread
From: Kyotaro Horiguchi @ 2026-06-02 05:29 UTC (permalink / raw)
To: mats.kindahl@gmail.com; +Cc: japinli@hotmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
At Tue, 2 Jun 2026 04:13:45 +0200, Mats Kindahl <mats.kindahl@gmail.com> wrote in
> If two servers go through the same sequence, e.g., start at the same
> timeline, does a promote, and write same length but different data
> (e.g., add a line to a table, but with different contents), they might
> end up with same TLI, same LSN, but different pg_promote calls, and
> different database contents, hence it is not possible to distinguish
> them.
Thanks for the explanation.
When I described UUIDs as somewhat heavyweight, I was thinking less
about the runtime overhead and more about operational convenience for
humans. Your clarification that the UUID only lives in the timeline
history file addresses most of the implementation concerns I had in
mind.
My earlier suggestion was based on the assumption that two independent
histories ending up with the same TLI and switchpoint LSN would be
rare enough to be ignored in practice. If the goal is to rule out that
possibility entirely rather than merely make it extremely unlikely,
then I can see why some additional identifier would be needed.
In that context, a UUID certainly seems like a viable option.
Regards.
--
Kyotaro Horiguchi
NTT Open Source Software Center
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-08 10:48 Andrey Borodin <x4mmm@yandex-team.ru>
parent: Mats Kindahl <mats.kindahl@gmail.com>
2 siblings, 1 reply; 35+ messages in thread
From: Andrey Borodin @ 2026-06-08 10:48 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
> On 30 Apr 2026, at 13:19, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>
> There is one scenario that I assume is known that TLC found, but does not seem to be fixed. It is a relatively rare case, but since the fix is quite easy, I thought I'd share it with you and get feedback.
Hi Mats,
Thanks for working on this. I think the problem is real, but I wonder if
adding a separate UUID to timeline history files is solving it one step
too late.
If two independent promotions manage to choose the same numeric TLI, then
we already have two different histories with the same timeline identifier.
Their history files will also have the same name. A UUID in the file lets
tools detect the mismatch afterwards, but it does not prevent the archive
namespace from containing two different meanings for the same TLI.
In normal deployments with a shared archive this should only be possible
when the history file is not visible to the other promoting server:
either there is no usable restore_command/shared archive, or there is a
race around publishing and observing the history file. In other words, TLI
allocation is not atomic, but it is intended to be coordinated through the
archive.
Maybe we should keep TimelineID as the actual branch identifier and make
that allocation harder to collide instead of adding a second identifier.
For example, when choosing a new TLI, add some randomness rather than just
using the next sequential value. That would make the race window much less
dangerous: two independent promotions would be extremely unlikely to
choose the same TLI, the history file names would remain distinct, and TLI
would keep its current role as the timeline identifier.
This also keeps the operational model simpler. TimelineID is already the
identifier exposed in WAL file names, history file names, logs, and
recovery configuration. If we add UUIDs, we effectively introduce another
identity for the same object, and tools then need to reason about both.
If instead we make TLI allocation less deterministic under races, the
existing model remains intact.
Does that framing make sense, or am I missing a case where duplicate TLIs
are unavoidable even with a shared archive and a less collision-prone
allocation scheme?
Best regards, Andrey Borodin.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-08 19:52 Zsolt Parragi <zsolt.parragi@percona.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
1 sibling, 1 reply; 35+ messages in thread
From: Zsolt Parragi @ 2026-06-08 19:52 UTC (permalink / raw)
To: pgsql-hackers@lists.postgresql.org
Hello
I know there's still ongoing discussion on the direction itself, but I
focused on just testing and looking at the latest patch, in case the
fix remains the same.
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
This is missing a MemoryContextSwitchTo before CopyErrorData, and
results in an assertion with debug builds.
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
This ignores the return value of rewind_parse_uuid, possibly writing
partial garbage to tluuid on incorrect input.
Also, it seems like that with this patch, pg rewind requires the
target's history file to be always there - is this an intended change?
If yes, then it should be at least mentioned somewhere.
On master:
exit 0
"source and target cluster are on the same timeline"
"no rewind required"
On patched rewind:
exit 1
error: could not open file
".../tgt/pg_wal/00000002.history" for reading: No such file or directory
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
In case this is a bug that should be backported, wouldn't this be an ABI break?
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64
unix_ts_ms, uint32 sub_ms);
Shouldn't these go before the endif?
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-06-21 09:09 Mats Kindahl <mats.kindahl@gmail.com>
parent: Andrey Borodin <x4mmm@yandex-team.ru>
0 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-06-21 09:09 UTC (permalink / raw)
To: Andrey Borodin <x4mmm@yandex-team.ru>; +Cc: pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
On 6/8/26 12:48, Andrey Borodin wrote:
>> On 30 Apr 2026, at 13:19, Mats Kindahl<mats.kindahl@gmail.com> wrote:
>>
>> There is one scenario that I assume is known that TLC found, but does not seem to be fixed. It is a relatively rare case, but since the fix is quite easy, I thought I'd share it with you and get feedback.
> Hi Mats,
Hi Andrey,
Thanks for looking at this.
> Thanks for working on this. I think the problem is real, but I wonder if
> adding a separate UUID to timeline history files is solving it one step
> too late.
>
> If two independent promotions manage to choose the same numeric TLI, then
> we already have two different histories with the same timeline identifier.
> Their history files will also have the same name. A UUID in the file lets
> tools detect the mismatch afterwards, but it does not prevent the archive
> namespace from containing two different meanings for the same TLI.
Yes, that is correct.
> In normal deployments with a shared archive this should only be possible
> when the history file is not visible to the other promoting server:
> either there is no usable restore_command/shared archive, or there is a
> race around publishing and observing the history file. In other words, TLI
> allocation is not atomic, but it is intended to be coordinated through the
> archive.
Yes, that is the ideal way it should work when you have a shared
archive. This works because you have a central authority that
synchronizes the timelines (in theory, not counting bugs).
> Maybe we should keep TimelineID as the actual branch identifier and make
> that allocation harder to collide instead of adding a second identifier.
> For example, when choosing a new TLI, add some randomness rather than just
> using the next sequential value.
> That would make the race window much less
> dangerous: two independent promotions would be extremely unlikely to
> choose the same TLI, the history file names would remain distinct, and TLI
> would keep its current role as the timeline identifier.
> This also keeps the operational model simpler. TimelineID is already the
> identifier exposed in WAL file names, history file names, logs, and
> recovery configuration. If we add UUIDs, we effectively introduce another
> identity for the same object, and tools then need to reason about both.
> If instead we make TLI allocation less deterministic under races, the
> existing model remains intact.
>
> Does that framing make sense, or am I missing a case where duplicate TLIs
> are unavoidable even with a shared archive and a less collision-prone
> allocation scheme?
I considered using some random increment of the TLI in the manner you
describe but there are some issues that makes this solution more
complicated from an operational perspective:
* If you skip some TLIs (in the sense pick a TLI that is "random but
larger"), then it is not clear what the relation between them are.
o The history files contain the complete linkage of the timelines,
so that is covered, but the naming would be strange.
+ For example, if you have history files 1, 5, 7, and 8, then
these can all belong to different timelines, (except 1), or
be a single timeline and it is hard to understand which one
without looking through the files.
o With more promotions, the relation becomes even more strange,
and the risk of collisions increases. (For example, imagine one
timeline with 1, 5, 7, 8, 11, and one timeline that forks off 1.
Then any increment of 4, 6, 7, or 10 will result in a collision.)
* To actually reduce the risk significantly, you need to have a very
wide range of the added randomness. Taking a smaller number is
easier to work with, but then you need to handle that some timelines
can collide in some manner.
* Normally, the history file with the highest number will be the only
relevant one. With this approach, you have to check the contents of
the files to understand which ones are relevant, which increases the
operational burden.
In contrast, if you use an UUID in this manner.
* Adding an UUID does not require a central coordinator and is not
likely to collide (on the level "impossible to collide") and is very
straightforward to add. It also comes with a low risk since the
places in the code that requires changes are very few and not likely
to have unexpected consequences elsewhere. This works both with and
without a shared archive.
* Normally, a shared archive should only contain a single timeline.
Anything else is an anomaly and should be corrected.
* I think it is still necessary to handle the case where you do not
have a shared archive; it would be an odd limitation to say that
promote only works if you have a shared archive
* The UUID still serves a purpose in capturing a situation where
things have gone wrong. Think of the UUID as similar to a "checksum"
safety and an extra precaution to prevent things from going wrong.
In short, I think the operational issues with random increment of the
history file number is worse, not better, and we should deal with the
name collisions correctly for shared archives instead. There is an issue
in that it need to work even in the case where you have a promotion that
generates a new UUID but the correct history file exists (reported in
the other message) that I will look into.
Best wishes,
Mats Kindahl
> Best regards, Andrey Borodin.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-17 15:59 Japin Li <japinli@hotmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 2 replies; 35+ messages in thread
From: Japin Li @ 2026-07-17 15:59 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: Andrey Borodin <x4mmm@yandex-team.ru>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
Hi, all
On Sun, 21 Jun 2026 at 11:09, Mats Kindahl <mats.kindahl@gmail.com> wrote:
> On 6/8/26 12:48, Andrey Borodin wrote:
>
> On 30 Apr 2026, at 13:19, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>
> There is one scenario that I assume is known that TLC found, but does not seem to be fixed. It is a relatively rare case, but since the fix is quite easy, I thought I'd share it with you and get feedback.
>
> Hi Mats,
>
> Hi Andrey,
>
> Thanks for looking at this.
>
> Thanks for working on this. I think the problem is real, but I wonder if
> adding a separate UUID to timeline history files is solving it one step
> too late.
>
> If two independent promotions manage to choose the same numeric TLI, then
> we already have two different histories with the same timeline identifier.
> Their history files will also have the same name. A UUID in the file lets
> tools detect the mismatch afterwards, but it does not prevent the archive
> namespace from containing two different meanings for the same TLI.
>
> Yes, that is correct.
>
> In normal deployments with a shared archive this should only be possible
> when the history file is not visible to the other promoting server:
> either there is no usable restore_command/shared archive, or there is a
> race around publishing and observing the history file. In other words, TLI
> allocation is not atomic, but it is intended to be coordinated through the
> archive.
>
> Yes, that is the ideal way it should work when you have a shared archive. This works because you have a central authority
> that synchronizes the timelines (in theory, not counting bugs).
>
> Maybe we should keep TimelineID as the actual branch identifier and make
> that allocation harder to collide instead of adding a second identifier.
> For example, when choosing a new TLI, add some randomness rather than just
> using the next sequential value.
>
> That would make the race window much less
> dangerous: two independent promotions would be extremely unlikely to
> choose the same TLI, the history file names would remain distinct, and TLI
> would keep its current role as the timeline identifier.
> This also keeps the operational model simpler. TimelineID is already the
> identifier exposed in WAL file names, history file names, logs, and
> recovery configuration. If we add UUIDs, we effectively introduce another
> identity for the same object, and tools then need to reason about both.
> If instead we make TLI allocation less deterministic under races, the
> existing model remains intact.
>
> Does that framing make sense, or am I missing a case where duplicate TLIs
> are unavoidable even with a shared archive and a less collision-prone
> allocation scheme?
>
> I considered using some random increment of the TLI in the manner you describe but there are some issues that makes this
> solution more complicated from an operational perspective:
>
> * If you skip some TLIs (in the sense pick a TLI that is "random but larger"), then it is not clear what the relation
> between them are.
>
> * The history files contain the complete linkage of the timelines, so that is covered, but the naming would be strange.
>
> * For example, if you have history files 1, 5, 7, and 8, then these can all belong to different timelines, (except 1), or
> be a single timeline and it is hard to understand which one without looking through the files.
>
> * With more promotions, the relation becomes even more strange, and the risk of collisions increases. (For example,
> imagine one timeline with 1, 5, 7, 8, 11, and one timeline that forks off 1. Then any increment of 4, 6, 7, or 10 will
> result in a collision.)
>
> * To actually reduce the risk significantly, you need to have a very wide range of the added randomness. Taking a smaller
> number is easier to work with, but then you need to handle that some timelines can collide in some manner.
> * Normally, the history file with the highest number will be the only relevant one. With this approach, you have to check
> the contents of the files to understand which ones are relevant, which increases the operational burden.
>
> In contrast, if you use an UUID in this manner.
>
> * Adding an UUID does not require a central coordinator and is not likely to collide (on the level "impossible to
> collide") and is very straightforward to add. It also comes with a low risk since the places in the code that requires
> changes are very few and not likely to have unexpected consequences elsewhere. This works both with and without a
> shared archive.
> * Normally, a shared archive should only contain a single timeline. Anything else is an anomaly and should be corrected.
> * I think it is still necessary to handle the case where you do not have a shared archive; it would be an odd limitation
> to say that promote only works if you have a shared archive
> * The UUID still serves a purpose in capturing a situation where things have gone wrong. Think of the UUID as similar to
> a "checksum" safety and an extra precaution to prevent things from going wrong.
>
> In short, I think the operational issues with random increment of the history file number is worse, not better, and we
> should deal with the name collisions correctly for shared archives instead. There is an issue in that it need to work
> even in the case where you have a promotion that generates a new UUID but the correct history file exists (reported in
> the other message) that I will look into.
>
I would like to know the current status of this patch. I have encountered the
same issue in practice, and I think the proposed solution is reasonable.
I found that the v6 patch does not apply cleanly to the current master (1f414035135)
because commit 7f77b2a89bd4 changed the parameter type of writeTimeLineHistory().
I've rebased the patch and attached v7.
> Best wishes,
> Mats Kindahl
>
> Best regards, Andrey Borodin.
--
Regards,
Japin Li
ChengDu WenWu Information Technology Co., Ltd.
Attachments:
[text/x-patch] v7-0001-pg_rewind-use-UUIDs-to-detect-independent-same-TL.patch (37.8K, ../../SY7PR01MB10921220B1B7054260BC8031FB6C62@SY7PR01MB10921.ausprd01.prod.outlook.com/2-v7-0001-pg_rewind-use-UUIDs-to-detect-independent-same-TL.patch)
download | inline diff:
From 71ef2382f00dec58ceef7fde614afe433d7ae19b Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: [PATCH v7] pg_rewind: use UUIDs to detect independent same-TLI
promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Two new tests in t/005_same_timeline.pl cover both detection paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
---
src/backend/access/transam/timeline.c | 77 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 104 ++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 362 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 47 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 614 insertions(+), 22 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index d80c8ffe0a7..237511b7521 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,45 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata = CopyErrorData();
+
+ FlushErrorState();
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +238,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +337,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
* Currently this is only used at the end recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, const char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
ssize_t nbytes;
@@ -398,13 +453,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index f8b939853e9..fde491bff5f 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6376,6 +6377,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6406,8 +6410,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index a9edceabab0..c153131e9f5 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -89,7 +89,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -616,6 +616,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -632,10 +640,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 2e86fd158d0..ffb12b1ba4a 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by matchAndFetchTimelines().
+ */
+typedef struct TimeLineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimeLineHistoriesData;
+
+typedef TimeLineHistoriesData *TimeLineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimeLineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimeLineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -374,10 +391,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -391,8 +419,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -876,7 +906,7 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
*/
if (tli == 1)
{
- history = pg_malloc_object(TimeLineHistoryEntry);
+ history = pg_malloc0_object(TimeLineHistoryEntry);
history->tli = tli;
history->begin = history->end = InvalidXLogRecPtr;
*nentries = 1;
@@ -922,6 +952,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimeLineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -943,12 +1023,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..2360c3df1d0 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,364 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+sub rewind_node
+{
+ my ($target, $source, $label, @extra_args) = @_;
+ $source->stop;
+ $target->stop;
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start;
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b2 both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+if ($PostgreSQL::Test::Utils::windows_os)
+{
+ $node_x_waldir =~ s{\\}{\\\\}g;
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'copy "$node_x_waldir\\\\%f" "%p"'\n));
+}
+else
+{
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'cp "$node_x_waldir/%f" "%p"'\n));
+}
+
+rewind_node($node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ '--restore-target-wal');
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..b6500606b27 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +98,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +108,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +132,14 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ rewind_parse_uuid(uuid_str, &entry->tluuid);
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +163,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 3aee3419a5c..3b94c4f0a4d 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, const char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, const char *content, size_t size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index be718993401..588094090c8 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..6839de2e0b2 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -38,5 +42,9 @@ DatumGetUUIDP(Datum X)
}
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+#endif /* !FRONTEND */
+
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
#endif /* UUID_H */
--
2.53.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-26 05:11 Mats Kindahl <mats.kindahl@gmail.com>
parent: Zsolt Parragi <zsolt.parragi@percona.com>
0 siblings, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-07-26 05:11 UTC (permalink / raw)
To: Zsolt Parragi <zsolt.parragi@percona.com>; pgsql-hackers@lists.postgresql.org
Hello Zsolt,
Thank you for reviewing this and sorry for the delay. I have attached a
new version with the issues you pointed to handled. See comments inline
below.
Best wishes,
Mats Kindahl
On 6/8/26 21:52, Zsolt Parragi wrote:
> Hello
>
> I know there's still ongoing discussion on the direction itself, but I
> focused on just testing and looking at the latest patch, in case the
> fix remains the same.
>
> + PG_CATCH();
> + {
> + ErrorData *edata = CopyErrorData();
> +
> + FlushErrorState();
> + ereport(FATAL,
> + errmsg("invalid UUID in history file \"%s\"", path),
> + errdetail("%s", edata->message));
> + }
>
> This is missing a MemoryContextSwitchTo before CopyErrorData, and
> results in an assertion with debug builds.
>
> + memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
> + if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
> + rewind_parse_uuid(uuid_str, &entry->tluuid);
>
> This ignores the return value of rewind_parse_uuid, possibly writing
> partial garbage to tluuid on incorrect input.
>
> Also, it seems like that with this patch, pg rewind requires the
> target's history file to be always there - is this an intended change?
> If yes, then it should be at least mentioned somewhere.
>
> On master:
> exit 0
> "source and target cluster are on the same timeline"
> "no rewind required"
>
> On patched rewind:
> exit 1
> error: could not open file
> ".../tgt/pg_wal/00000002.history" for reading: No such file or directory
That was not the intention, so I added a check for a target history file
to the code and a test that checks the behavior without a timeline
history file and compared that with the behavior before the change.
>
>
> writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
> + const pg_uuid_t *newTLUUID,
> XLogRecPtr switchpoint, char *reason)
>
> In case this is a bug that should be backported, wouldn't this be an ABI break?
>
>
> +#endif /* !FRONTEND */
> +
> +extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
> +extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64
> unix_ts_ms, uint32 sub_ms);
>
> Shouldn't these go before the endif?
I thought the generate_uuidv7 should be possible to use for a frontend,
but looking at how it is written and linked, that is not possible to I
moved the #endif.
Attachments:
[text/x-patch] v7.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (43.9K, ../../8683af69-28af-4a2d-a1db-aa1447b02446@gmail.com/2-v7.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 68d321732f5a82ddbe0a13eab03e9914bfbebba4 Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Before this commit, Postgres tolerated the target having no local copy
of the history file for its own current TLI. This could happen if, for
example, a standby that started streaming a timeline it never itself
switched to, hence had no reason to write or fetch a copy of that file.
This behaviour is retained.
Tests in t/005_same_timeline.pl cover these paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
The third covers the missing-history-file fallback: two standbys
independently promote to TLI2, the target's own TLI2 history file is
then removed, and pg_rewind is checked to no longer abort because of it.
---
src/backend/access/transam/timeline.c | 84 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 159 +++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 459 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 50 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 762 insertions(+), 36 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index 68e5f692d26..dee579c039b 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,50 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ MemoryContext oldcontext = CurrentMemoryContext;
+
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata;
+
+ MemoryContextSwitchTo(oldcontext);
+ edata = CopyErrorData();
+ FlushErrorState();
+
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +243,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +342,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
int nbytes;
@@ -398,13 +458,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index d69d03b2ef3..3176b74c025 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6377,6 +6378,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6407,8 +6411,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index 6ee3752ac78..f1dc0196cd8 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -72,7 +72,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -581,6 +581,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -597,10 +605,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index 9d745d4b25b..069c5bb5a90 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by matchAndFetchTimelines().
+ */
+typedef struct TimeLineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimeLineHistoriesData;
+
+typedef TimeLineHistoriesData * TimeLineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -53,6 +66,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimeLineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -141,6 +157,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimeLineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -372,10 +389,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -389,8 +417,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -860,40 +890,69 @@ MinXLogRecPtr(XLogRecPtr a, XLogRecPtr b)
return Min(a, b);
}
+static bool
+file_exists(const char *name)
+{
+ struct stat st;
+
+ Assert(name != NULL);
+
+ if (stat(name, &st) == 0)
+ return !S_ISDIR(st.st_mode);
+ else if (!(errno == ENOENT || errno == ENOTDIR || errno == EACCES))
+ pg_fatal("could not stat file \"%s\": %m", name);
+
+ return false;
+
+}
+
/*
* Retrieve timeline history for the source or target system.
*/
static TimeLineHistoryEntry *
getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
{
- TimeLineHistoryEntry *history;
+ TimeLineHistoryEntry *history = NULL;
/*
* Timeline 1 does not have a history file, so there is no need to check
* and fake an entry with infinite start and end positions.
*/
- if (tli == 1)
- {
- history = pg_malloc_object(TimeLineHistoryEntry);
- history->tli = tli;
- history->begin = history->end = InvalidXLogRecPtr;
- *nentries = 1;
- }
- else
+ if (tli != 1)
{
char path[MAXPGPATH];
- char *histfile;
+ char *histfile = NULL;
TLHistoryFilePath(path, tli);
- /* Get history file from appropriate source */
+ /* Get history file from appropriate source, tolerating its absence */
if (is_source)
histfile = source->fetch_file(source, path, NULL);
else
- histfile = slurpFile(datadir_target, path, NULL);
+ {
+ char fullpath[MAXPGPATH];
+
+ snprintf(fullpath, sizeof(fullpath), "%s/%s", datadir_target, path);
+ if (file_exists(fullpath))
+ histfile = slurpFile(datadir_target, path, NULL);
+ }
+
+ if (histfile != NULL)
+ {
+ history = rewind_parseTimeLineHistory(histfile, tli, nentries);
+ pg_free(histfile);
+ }
+ }
- history = rewind_parseTimeLineHistory(histfile, tli, nentries);
- pg_free(histfile);
+ /*
+ * If no history file entry was created, we create a "zero" entry.
+ */
+ if (history == NULL)
+ {
+ history = pg_malloc0_object(TimeLineHistoryEntry);
+ history->tli = tli;
+ history->begin = history->end = InvalidXLogRecPtr;
+ *nentries = 1;
}
/* In debugging mode, print what we read */
@@ -920,6 +979,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimeLineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -941,12 +1050,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..b1797f78881 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,461 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+#
+# If the last element of @extra_args is a coderef, it is not passed to
+# pg_rewind: it is run after pg_rewind finishes but before the target is
+# restarted, e.g. to undo a test's own tampering with the target's data
+# directory that pg_rewind itself doesn't need to be aware of.
+sub rewind_node
+{
+ my ($target, $source, $label, %opts) = @_;
+ my @extra_args;
+
+ $source->stop;
+ $target->stop;
+
+ push @extra_args, '--restore-target-wal' if $opts{'-restorewal'};
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start unless $opts{'-norestart'};
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Remove a node's TLI history file entirely, simulating a node that has no local
+# record of its own current-TLI promotion, for example, because it started
+# streaming a timeline it never itself switched to. This means that it never had
+# a reason to write (or fetch) a local copy of that history file. Returns the
+# removed file's content, so it can be restored with restore_tli_history().
+#
+# Note this simulation is imperfect: unlike a node that never had that reason,
+# our node genuinely did promote and so still has the earlier TLI's real WAL
+# segments on disk. Without the history file linking them, the server itself can
+# no longer make sense of its own pg_wal directory and will fail to start --
+# restore_tli_history() must be called before restarting the node.
+sub remove_tli_history
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ local $/;
+ my $content = <$fh>;
+ close $fh;
+ unlink($histfile) or die "cannot remove $histfile: $!";
+ return $content;
+}
+
+# Restore a TLI history file previously removed by remove_tli_history().
+sub restore_tli_history
+{
+ my ($node, $tli, $content) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '>', $histfile) or die "cannot write $histfile: $!";
+ print $fh $content;
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b2 both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+if ($PostgreSQL::Test::Utils::windows_os)
+{
+ $node_x_waldir =~ s{\\}{\\\\}g;
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'copy "$node_x_waldir\\\\%f" "%p"'\n));
+}
+else
+{
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'cp "$node_x_waldir/%f" "%p"'\n));
+}
+
+rewind_node(
+ $node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ -restorewal => 1);
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
+# Test that pg_rewind does not fail when the target has no local copy of its own
+# current TLI's history file, for example, because it started streaming that TLI
+# from a primary that had already promoted, without ever itself performing a
+# promotion or otherwise obtaining a local copy of the file.
+#
+# Without that file, pg_rewind cannot read the promotion UUID for the shared TLI
+# on the target side, and so cannot tell whether the two clusters are the same
+# promotion or two independent ones.
+#
+# We check that no rewind does actually take place, not just that the error
+# message is correct. For that reason, we create two divergent nodes for the
+# test, remove the history file, and then attempt a rewind. If the code works
+# correctly, the node should not be rewound.
+my $node_origin5 = setup_origin('origin5');
+my ($node_g, $node_h) =
+ setup_standbys_from_origin($node_origin5, 'node_g', 'node_h');
+
+sync_standbys_with_origin($node_origin5, $node_g, $node_h);
+$node_origin5->stop;
+
+$node_g->promote;
+$node_h->promote;
+
+write_record($node_g, 'in G');
+write_record($node_h, 'in H');
+
+my $saved_history = remove_tli_history($node_g, 2);
+
+rewind_node(
+ $node_g, $node_h,
+ 'pg_rewind does not fail when target history file is missing',
+ -norestart => 1);
+
+# Restore the history file of node_g before it gets restarted. We are testing
+# that pg_rewind can handle a missing history file, but unlike the real-world
+# case this simulates, the pg_wal directory of node_g still has its own genuine
+# TLI 1 segments on disk and the server cannot start back up without the file
+# linking them to TLI 2.
+restore_tli_history($node_g, 2, $saved_history);
+
+$node_g->restart;
+
+my $result5 =
+ $node_g->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is( $result5,
+ "in G\norigin5",
+ 'missing history file prevents divergence detection: target data left unchanged'
+);
+
+$node_g->teardown_node;
+$node_h->teardown_node;
+$node_origin5->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index dda06eaa0bc..7bfadc21bdd 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,9 +9,40 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -48,6 +79,9 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
uint32 switchpoint_hi;
uint32 switchpoint_lo;
int nfields;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
+ pg_uuid_t buf;
+
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -66,7 +100,8 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields = sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi,
+ &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -75,7 +110,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
pg_log_error_detail("Expected a numeric timeline ID.");
exit(1);
}
- if (nfields != 3)
+ if (nfields < 3)
{
pg_log_error("syntax error in history file: %s", fline);
pg_log_error_detail("Expected a write-ahead log switchpoint location.");
@@ -99,7 +134,15 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4; its first word is much shorter than UUID_STR_LEN
+ * so the length check safely distinguishes old from new format.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ if (rewind_parse_uuid(uuid_str, &buf))
+ memcpy(&entry->tluuid, &buf, sizeof(pg_uuid_t));
}
if (entries && targetTLI <= lasttli)
@@ -123,6 +166,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 97f1d619c35..cdd642c94f0 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, char *content, int size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index 55663e6f4af..20a2f345fd3 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..47a6af0ab3c 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -39,4 +43,8 @@ DatumGetUUIDP(Datum X)
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
+
+#endif /* !FRONTEND */
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-26 09:57 Tatsuya Kawata <kawatatatsuya0913@gmail.com>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Tatsuya Kawata @ 2026-07-26 09:57 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: Zsolt Parragi <zsolt.parragi@percona.com>; pgsql-hackers@lists.postgresql.org
Hi Mats-san, Zsolt-san,
Thanks -- I went through both v7 and the new version.
> + PG_CATCH();
> + {
> + ErrorData *edata = CopyErrorData();
> +
> + FlushErrorState();
> + ereport(FATAL,
> + errmsg("invalid UUID in
history file \"%s\"", path),
> + errdetail("%s",
edata->message));
> + }
>
> This is missing a MemoryContextSwitchTo before CopyErrorData, and
> results in an assertion with debug builds.
> Thank you for reviewing this and sorry for the delay. I have attached a
> new version with the issues you pointed to handled. See comments inline
> below.
The context-switch
fix in readTimeLineHistory() (restoring the caller's context before
CopyErrorData()) looks correct to me.
One note: the original problem was not only a debug-build assertion. On
non-assert builds CopyErrorData() allocates the ErrorData in ErrorContext,
FlushErrorState() then frees it, and the following
errdetail("%s", edata->message) reads freed memory -- a use-after-free that
can crash a production server, not just trip an Assert(). Your fix already
covers this; I'm just sharing it since it bears on the severity.
One minor point: on an invalid UUID the backend FATALs while the frontend
(pg_rewind) silently treats it as "unknown" (all-zero) -- probably
intentional, just flagging it. And should you ever want to drop the
PG_TRY/PG_CATCH here, uuid_in supports soft errors, so a
DirectInputFunctionCallSafe() call with an ErrorSaveContext would avoid
CopyErrorData()/FlushErrorState() and the context switch entirely -- i.e.
it removes the very handling that had to be fixed here, so this class of
mistake can't recur. The current fix is correct and minimal, so this is
purely optional.
Regards,
Tatsuya Kawata
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-26 23:36 Michael Paquier <michael@paquier.xyz>
parent: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
3 siblings, 3 replies; 35+ messages in thread
From: Michael Paquier @ 2026-07-26 23:36 UTC (permalink / raw)
To: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: japinli@hotmail.com; mats.kindahl@gmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
> I wonder whether strengthening the history-based matching would be
> sufficient instead. If timelines with the same TLI but different
> histories can be treated as distinct and pg_rewind continues walking
> the history chain until it finds a common ancestor, that seems like a
> fairly natural fit with the existing timeline model.
Yeah, perhaps there is something that we could do here better in
pg_rewind in terms of TLI history. A second software layer to ensure
TLI unicity feels just like a shortcut: we already have a LSN.
> UUIDs would certainly make identification straightforward, although
> they would also introduce longer identifiers that are a bit less
> convenient for humans to work with. My initial thought is that it may
> be worth exploring how far we can get with the existing history
> information before introducing a new identifier.
This. For this reason, I am not convinced that this proposal is
neither acceptable or something that we need to do at all.
Why should we bear the burden of introducing a second level of
timeline identification knowing that by design we have to rely on a
*single* archive location for a new timeline selection when a standby
is triggered for promotion? My opinion is that this is trying to fix
a problem for something that it not actually a problem. If you play
with HA scenarios where there is a risk of two standbys reusing the
same timeline number, just don't do that. We've historically claimed
that the archive strategy is wrong if your deployments cannot
guarantee a unique TLI assignment. A new sub-identification system
will not provide more guarantees.
--
Michael
Attachments:
[application/pgp-signature] signature.asc (832B, ../../amaZ6bN0pJM75OEs@paquier.xyz/2-signature.asc)
download
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-29 11:51 Ants Aasma <ants.aasma@cybertec.at>
parent: Michael Paquier <michael@paquier.xyz>
2 siblings, 1 reply; 35+ messages in thread
From: Ants Aasma @ 2026-07-29 11:51 UTC (permalink / raw)
To: Michael Paquier <michael@paquier.xyz>; +Cc: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; japinli@hotmail.com; mats.kindahl@gmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On Mon, 27 Jul 2026 at 02:36, Michael Paquier <michael@paquier.xyz> wrote:
>
> On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
> > I wonder whether strengthening the history-based matching would be
> > sufficient instead. If timelines with the same TLI but different
> > histories can be treated as distinct and pg_rewind continues walking
> > the history chain until it finds a common ancestor, that seems like a
> > fairly natural fit with the existing timeline model.
>
> Yeah, perhaps there is something that we could do here better in
> pg_rewind in terms of TLI history. A second software layer to ensure
> TLI unicity feels just like a shortcut: we already have a LSN.
The LSN can easily match for two promotions.
> > UUIDs would certainly make identification straightforward, although
> > they would also introduce longer identifiers that are a bit less
> > convenient for humans to work with. My initial thought is that it may
> > be worth exploring how far we can get with the existing history
> > information before introducing a new identifier.
>
> This. For this reason, I am not convinced that this proposal is
> neither acceptable or something that we need to do at all.
>
> Why should we bear the burden of introducing a second level of
> timeline identification knowing that by design we have to rely on a
> *single* archive location for a new timeline selection when a standby
> is triggered for promotion? My opinion is that this is trying to fix
> a problem for something that it not actually a problem. If you play
> with HA scenarios where there is a risk of two standbys reusing the
> same timeline number, just don't do that. We've historically claimed
> that the archive strategy is wrong if your deployments cannot
> guarantee a unique TLI assignment. A new sub-identification system
> will not provide more guarantees.
The trouble is that with the current archiving model it is impossible to
ensure a unique TLI assignment. Restore command checks for existence, the
first identifier that gives an error will get picked, the promotion
finished and then later it is asynchronously archived. If there is a
failure between promotion and archiving you will get multiple nodes on the
same timeline. There's also the issue that restore command has no way of
distinguishing whether the archive is unavailable or the requested file is
missing.
Having two failures so close by might sound improbable, but my experience
shows that failures are very often correlated. E.g. a gray failure of
networked storage can cause systems to become unresponsive to the extent of
being indistinguishable from dead, and this often happens in a rolling
manner. Primary fails, secondary promotes and then immediately fails
itself. I have seen this happen almost daily when someone misconfigures
their backups to happen at the same time across their fleet.
However I am not sure if this proposal is going far enough. The goal here
is to uniquely identify transaction log to originate from a specific
postgres instance running as a primary. It is reasonable to expect that HA
systems ensure that for each instance running as primary there exists a
single promote call. But not really much else. This proposal only handles
the history file uniqueness in pg_rewind. Unless the archive command is
very carefully managed and the archive itself is fully consistent, it is
also possible to get WAL files in the archive from different primaries but
with the same TLI. If things go well this might just result in replication
failing, but if things align in the wrong way it can also cause silent data
corruption. This is still too fragile for my taste.
One straightforward solution is to replace the whole TLI with a UUID and
invent some new method of determining "latest" timeline. This has the
obvious problem of breaking almost all of the existing tooling. But a less
invasive way to at least detect issues and give backup tools ways to solve
the problem would be to use this UUID proposal and also include it in
XLogLongPageHeaderData. Recovery can then validate that the WAL file is
from the correct timeline.
Regards,
Ants Aasma
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-07-29 12:55 Andreas Karlsson <andreas@proxel.se>
parent: Michael Paquier <michael@paquier.xyz>
2 siblings, 1 reply; 35+ messages in thread
From: Andreas Karlsson @ 2026-07-29 12:55 UTC (permalink / raw)
To: Michael Paquier <michael@paquier.xyz>; Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: japinli@hotmail.com; mats.kindahl@gmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 7/27/26 01:36, Michael Paquier wrote:
> On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
>> I wonder whether strengthening the history-based matching would be
>> sufficient instead. If timelines with the same TLI but different
>> histories can be treated as distinct and pg_rewind continues walking
>> the history chain until it finds a common ancestor, that seems like a
>> fairly natural fit with the existing timeline model.
>
> Yeah, perhaps there is something that we could do here better in
> pg_rewind in terms of TLI history. A second software layer to ensure
> TLI unicity feels just like a shortcut: we already have a LSN.
As Ants said, the LSN can easily be the same in both.
>> UUIDs would certainly make identification straightforward, although
>> they would also introduce longer identifiers that are a bit less
>> convenient for humans to work with. My initial thought is that it may
>> be worth exploring how far we can get with the existing history
>> information before introducing a new identifier.
>
> This. For this reason, I am not convinced that this proposal is
> neither acceptable or something that we need to do at all.
>
> Why should we bear the burden of introducing a second level of
> timeline identification knowing that by design we have to rely on a
> *single* archive location for a new timeline selection when a standby
> is triggered for promotion? My opinion is that this is trying to fix
> a problem for something that it not actually a problem. If you play
> with HA scenarios where there is a risk of two standbys reusing the
> same timeline number, just don't do that. We've historically claimed
> that the archive strategy is wrong if your deployments cannot
> guarantee a unique TLI assignment. A new sub-identification system
> will not provide more guarantees.
Having had to debug issues with pg_rewind and timelines in our
Kubernetes Operator these UUIDs would probably have helped me a lot.
Maybe the proposed fix is wrong (maybe like Ants says it does not got
far enough) but from personal experience I disagree with that there is
no problem here.
I will try to find time to play around with the patch and see if it
would have helped us.
--
Andreas Karlsson
Percona
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-12 05:05 Mats Kindahl <mats.kindahl@gmail.com>
parent: Kyotaro Horiguchi <horikyota.ntt@gmail.com>
3 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-08-12 05:05 UTC (permalink / raw)
To: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; japinli@hotmail.com; +Cc: suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 6/1/26 08:30, Kyotaro Horiguchi wrote:
> Sorry, I only just noticed this thread.
>
> I may be missing something, but UUID feels somewhat heavyweight to me
> for this problem.
>
> I wonder whether strengthening the history-based matching would be
> sufficient instead. If timelines with the same TLI but different
> histories can be treated as distinct and pg_rewind continues walking
> the history chain until it finds a common ancestor, that seems like a
> fairly natural fit with the existing timeline model.
Unfortunately, I do not think that will work. For the normally
problematic case, the two competing TLIs will have the same history. I
added extra tests to support a divergence further back, but I find this
a rare case.
> UUIDs would certainly make identification straightforward, although
> they would also introduce longer identifiers that are a bit less
> convenient for humans to work with. My initial thought is that it may
> be worth exploring how far we can get with the existing history
> information before introducing a new identifier.
Well, although they are written to the file, the normal TLI is just an
extra defense, not a replacement for the TLI.
Best wishes,
Mats Kindahl
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-12 06:14 Mats Kindahl <mats.kindahl@gmail.com>
parent: Michael Paquier <michael@paquier.xyz>
2 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-08-12 06:14 UTC (permalink / raw)
To: Michael Paquier <michael@paquier.xyz>; Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: japinli@hotmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 7/27/26 01:36, Michael Paquier wrote:
> On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
>> I wonder whether strengthening the history-based matching would be
>> sufficient instead. If timelines with the same TLI but different
>> histories can be treated as distinct and pg_rewind continues walking
>> the history chain until it finds a common ancestor, that seems like a
>> fairly natural fit with the existing timeline model.
> Yeah, perhaps there is something that we could do here better in
> pg_rewind in terms of TLI history. A second software layer to ensure
> TLI unicity feels just like a shortcut: we already have a LSN.
>
>> UUIDs would certainly make identification straightforward, although
>> they would also introduce longer identifiers that are a bit less
>> convenient for humans to work with. My initial thought is that it may
>> be worth exploring how far we can get with the existing history
>> information before introducing a new identifier.
> This. For this reason, I am not convinced that this proposal is
> neither acceptable or something that we need to do at all.
>
> Why should we bear the burden of introducing a second level of
> timeline identification knowing that by design we have to rely on a
> *single* archive location for a new timeline selection when a standby
> is triggered for promotion? My opinion is that this is trying to fix
> a problem for something that it not actually a problem. If you play
> with HA scenarios where there is a risk of two standbys reusing the
> same timeline number, just don't do that.
Even with a shared archive, it is impossible to ensure such things. It
is a normal consensus problem, and there are just to many ways that it
can go wrong. I think the UUID should not be looked on as a replacement
for the TLI but rather as an an extra protection against problems. It
will ensure that we cannot even accidentally pick the wrong "version" of
the timeline, but keep the simplicity of using TLI for anything else.
> We've historically claimed
> that the archive strategy is wrong if your deployments cannot
> guarantee a unique TLI assignment. A new sub-identification system
> will not provide more guarantees.
It will serve as an extra check to ensure that you do not pick the wrong
version of a timeline, similar to how checksums ensure that you are not
using a page that is corrupt.
Best wishes,
Mats Kindahl, Multigres team, Supabase
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-12 06:32 Mats Kindahl <mats.kindahl@gmail.com>
parent: Ants Aasma <ants.aasma@cybertec.at>
0 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-08-12 06:32 UTC (permalink / raw)
To: Ants Aasma <ants.aasma@cybertec.at>; Michael Paquier <michael@paquier.xyz>; +Cc: Kyotaro Horiguchi <horikyota.ntt@gmail.com>; japinli@hotmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 7/29/26 13:51, Ants Aasma wrote:
> On Mon, 27 Jul 2026 at 02:36, Michael Paquier <michael@paquier.xyz> wrote:
> >
> > On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
> > > I wonder whether strengthening the history-based matching would be
> > > sufficient instead. If timelines with the same TLI but different
> > > histories can be treated as distinct and pg_rewind continues walking
> > > the history chain until it finds a common ancestor, that seems like a
> > > fairly natural fit with the existing timeline model.
> >
> > Yeah, perhaps there is something that we could do here better in
> > pg_rewind in terms of TLI history. A second software layer to ensure
> > TLI unicity feels just like a shortcut: we already have a LSN.
>
> The LSN can easily match for two promotions.
>
> > > UUIDs would certainly make identification straightforward, although
> > > they would also introduce longer identifiers that are a bit less
> > > convenient for humans to work with. My initial thought is that it may
> > > be worth exploring how far we can get with the existing history
> > > information before introducing a new identifier.
> >
> > This. For this reason, I am not convinced that this proposal is
> > neither acceptable or something that we need to do at all.
> >
> > Why should we bear the burden of introducing a second level of
> > timeline identification knowing that by design we have to rely on a
> > *single* archive location for a new timeline selection when a standby
> > is triggered for promotion? My opinion is that this is trying to fix
> > a problem for something that it not actually a problem. If you play
> > with HA scenarios where there is a risk of two standbys reusing the
> > same timeline number, just don't do that. We've historically claimed
> > that the archive strategy is wrong if your deployments cannot
> > guarantee a unique TLI assignment. A new sub-identification system
> > will not provide more guarantees.
>
> The trouble is that with the current archiving model it is impossible
> to ensure a unique TLI assignment. Restore command checks for
> existence, the first identifier that gives an error will get picked,
> the promotion finished and then later it is asynchronously archived.
> If there is a failure between promotion and archiving you will get
> multiple nodes on the same timeline. There's also the issue that
> restore command has no way of distinguishing whether the archive is
> unavailable or the requested file is missing.
>
> Having two failures so close by might sound improbable, but my
> experience shows that failures are very often correlated. E.g. a gray
> failure of networked storage can cause systems to become unresponsive
> to the extent of being indistinguishable from dead, and this often
> happens in a rolling manner. Primary fails, secondary promotes and
> then immediately fails itself. I have seen this happen almost daily
> when someone misconfigures their backups to happen at the same time
> across their fleet.
These are good examples of how things can go wrong, but even if they
were uncommon, having an extra check that you're not using a bad
timeline is beneficial. Also, failures like this have a tendency to
happen in bursts when something breaks, and having everything failing at
the same time will be problematic to deal with. Imagine you have a link
failure and a bunch of servers happen to pick different timelines and
make some changes before failing over again. Trying to sort out that
kind of situation would be a nightmare.
>
> However I am not sure if this proposal is going far enough. The goal
> here is to uniquely identify transaction log to originate from a
> specific postgres instance running as a primary. It is reasonable to
> expect that HA systems ensure that for each instance running as
> primary there exists a single promote call. But not really much else.
This is true, but it is a very difficult problem to ensure that you have
consensus on who should do that promote.
> This proposal only handles the history file uniqueness in pg_rewind.
> Unless the archive command is very carefully managed and the archive
> itself is fully consistent, it is also possible to get WAL files in
> the archive from different primaries but with the same TLI. If things
> go well this might just result in replication failing, but if things
> align in the wrong way it can also cause silent data corruption. This
> is still too fragile for my taste.
I'm not sure about the scenario you are thinking about here. You can
have different primaries that write to the archive, yes, but when you
restore they should come from one or the other. They should be
"internally consistent" in the sense that you are not mixing, for
example, different versions of WALs from different primaries. If you do
that, a recovery will be impossible anyway, so that would be a
different, but quite serious, problem. If you write different versions
of timelines with same TLI that should still work since the UUID would
ensure that you are not using the wrong version of the timeline.
It might be hard to *avoid* the problem entirely, but it should not be
possible to cause a data corruption (but I might misunderstand what you
had in mind).
>
> One straightforward solution is to replace the whole TLI with a UUID
> and invent some new method of determining "latest" timeline. This has
> the obvious problem of breaking almost all of the existing tooling.
> But a less invasive way to at least detect issues and give backup
> tools ways to solve the problem would be to use this UUID proposal and
> also include it in XLogLongPageHeaderData. Recovery can then validate
> that the WAL file is from the correct timeline.
I avoided making the UUID first class to not disrupt the existing
tooling too much, but that was only the reason. One of the earlier
versions also had the UUID in the WAL, but I skipped that since it was
quite intrusive and making modifications to the format of the WAL has
its own set of problems and can break things quite seriously. At least
that was my reasoning.
Best wishes,
Mats Kindahl, Multigres team, Supabase
>
> Regards,
> Ants Aasma
>
>
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-12 06:33 Mats Kindahl <mats.kindahl@gmail.com>
parent: Andreas Karlsson <andreas@proxel.se>
0 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-08-12 06:33 UTC (permalink / raw)
To: Andreas Karlsson <andreas@proxel.se>; Michael Paquier <michael@paquier.xyz>; Kyotaro Horiguchi <horikyota.ntt@gmail.com>; +Cc: japinli@hotmail.com; suryapoondla4@gmail.com; pgsql-hackers@lists.postgresql.org
On 7/29/26 14:55, Andreas Karlsson wrote:
> On 7/27/26 01:36, Michael Paquier wrote:
>> On Mon, Jun 01, 2026 at 03:30:58PM +0900, Kyotaro Horiguchi wrote:
>>> I wonder whether strengthening the history-based matching would be
>>> sufficient instead. If timelines with the same TLI but different
>>> histories can be treated as distinct and pg_rewind continues walking
>>> the history chain until it finds a common ancestor, that seems like a
>>> fairly natural fit with the existing timeline model.
>>
>> Yeah, perhaps there is something that we could do here better in
>> pg_rewind in terms of TLI history. A second software layer to ensure
>> TLI unicity feels just like a shortcut: we already have a LSN.
>
> As Ants said, the LSN can easily be the same in both.
>
>>> UUIDs would certainly make identification straightforward, although
>>> they would also introduce longer identifiers that are a bit less
>>> convenient for humans to work with. My initial thought is that it may
>>> be worth exploring how far we can get with the existing history
>>> information before introducing a new identifier.
>>
>> This. For this reason, I am not convinced that this proposal is
>> neither acceptable or something that we need to do at all.
>>
>> Why should we bear the burden of introducing a second level of
>> timeline identification knowing that by design we have to rely on a
>> *single* archive location for a new timeline selection when a standby
>> is triggered for promotion? My opinion is that this is trying to fix
>> a problem for something that it not actually a problem. If you play
>> with HA scenarios where there is a risk of two standbys reusing the
>> same timeline number, just don't do that. We've historically claimed
>> that the archive strategy is wrong if your deployments cannot
>> guarantee a unique TLI assignment. A new sub-identification system
>> will not provide more guarantees.
>
> Having had to debug issues with pg_rewind and timelines in our
> Kubernetes Operator these UUIDs would probably have helped me a lot.
> Maybe the proposed fix is wrong (maybe like Ants says it does not got
> far enough) but from personal experience I disagree with that there is
> no problem here.
>
> I will try to find time to play around with the patch and see if it
> would have helped us.
Thanks Andreas,
I would love to hear if this would have been a help in those scenarios.
Best wishes,
Mats Kindahl, Multigres team, Supabase
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-16 11:37 Mats Kindahl <mats.kindahl@gmail.com>
parent: Tatsuya Kawata <kawatatatsuya0913@gmail.com>
0 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-08-16 11:37 UTC (permalink / raw)
To: Tatsuya Kawata <kawatatatsuya0913@gmail.com>; +Cc: Zsolt Parragi <zsolt.parragi@percona.com>; pgsql-hackers@lists.postgresql.org
On 7/26/26 11:57, Tatsuya Kawata wrote:
> Hi Mats-san, Zsolt-san,
>
> Thanks -- I went through both v7 and the new version.
>
> > + PG_CATCH();
> > + {
> > + ErrorData *edata = CopyErrorData();
> > +
> > + FlushErrorState();
> > + ereport(FATAL,
> > + errmsg("invalid UUID in history file \"%s\"", path),
> > + errdetail("%s", edata->message));
> > + }
> >
> > This is missing a MemoryContextSwitchTo before CopyErrorData, and
> > results in an assertion with debug builds.
>
> > Thank you for reviewing this and sorry for the delay. I have attached a
> > new version with the issues you pointed to handled. See comments inline
> > below.
>
> The context-switch
> fix in readTimeLineHistory() (restoring the caller's context before
> CopyErrorData()) looks correct to me.
>
> One note: the original problem was not only a debug-build assertion. On
> non-assert builds CopyErrorData() allocates the ErrorData in ErrorContext,
> FlushErrorState() then frees it, and the following
> errdetail("%s", edata->message) reads freed memory -- a use-after-free
> that
> can crash a production server, not just trip an Assert(). Your fix already
> covers this; I'm just sharing it since it bears on the severity.
Got that. Assertions are just a way to trigger a potential problem
early. I did not assume this change was needed just to avoid the assertion.
> One minor point: on an invalid UUID the backend FATALs while the frontend
> (pg_rewind) silently treats it as "unknown" (all-zero) -- probably
> intentional, just flagging it.And should you ever want to drop the
> PG_TRY/PG_CATCH here, uuid_in supports soft errors, so a
> DirectInputFunctionCallSafe() call with an ErrorSaveContext would avoid
> CopyErrorData()/FlushErrorState() and the context switch entirely -- i.e.
> it removes the very handling that had to be fixed here, so this class of
> mistake can't recur. The current fix is correct and minimal, so this is
> purely optional.
Yes, I wanted to keep the UUID just as a final discriminator, after the
TLI, and keep the changes minimal.
Best wishes,
Mats Kindahl
>
> Regards,
> Tatsuya Kawata
>
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-16 11:54 Mats Kindahl <mats.kindahl@gmail.com>
parent: Japin Li <japinli@hotmail.com>
1 sibling, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-08-16 11:54 UTC (permalink / raw)
To: Japin Li <japinli@hotmail.com>; +Cc: Andrey Borodin <x4mmm@yandex-team.ru>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
Thank you Japin,
I double checked with the latest as well, just in case, but your patch
looks fine.
Best wishes,
Mats Kindahl
On 7/17/26 17:59, Japin Li wrote:
> Hi, all
>
> On Sun, 21 Jun 2026 at 11:09, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>> On 6/8/26 12:48, Andrey Borodin wrote:
>>
>> On 30 Apr 2026, at 13:19, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>>
>> There is one scenario that I assume is known that TLC found, but does not seem to be fixed. It is a relatively rare case, but since the fix is quite easy, I thought I'd share it with you and get feedback.
>>
>> Hi Mats,
>>
>> Hi Andrey,
>>
>> Thanks for looking at this.
>>
>> Thanks for working on this. I think the problem is real, but I wonder if
>> adding a separate UUID to timeline history files is solving it one step
>> too late.
>>
>> If two independent promotions manage to choose the same numeric TLI, then
>> we already have two different histories with the same timeline identifier.
>> Their history files will also have the same name. A UUID in the file lets
>> tools detect the mismatch afterwards, but it does not prevent the archive
>> namespace from containing two different meanings for the same TLI.
>>
>> Yes, that is correct.
>>
>> In normal deployments with a shared archive this should only be possible
>> when the history file is not visible to the other promoting server:
>> either there is no usable restore_command/shared archive, or there is a
>> race around publishing and observing the history file. In other words, TLI
>> allocation is not atomic, but it is intended to be coordinated through the
>> archive.
>>
>> Yes, that is the ideal way it should work when you have a shared archive. This works because you have a central authority
>> that synchronizes the timelines (in theory, not counting bugs).
>>
>> Maybe we should keep TimelineID as the actual branch identifier and make
>> that allocation harder to collide instead of adding a second identifier.
>> For example, when choosing a new TLI, add some randomness rather than just
>> using the next sequential value.
>>
>> That would make the race window much less
>> dangerous: two independent promotions would be extremely unlikely to
>> choose the same TLI, the history file names would remain distinct, and TLI
>> would keep its current role as the timeline identifier.
>> This also keeps the operational model simpler. TimelineID is already the
>> identifier exposed in WAL file names, history file names, logs, and
>> recovery configuration. If we add UUIDs, we effectively introduce another
>> identity for the same object, and tools then need to reason about both.
>> If instead we make TLI allocation less deterministic under races, the
>> existing model remains intact.
>>
>> Does that framing make sense, or am I missing a case where duplicate TLIs
>> are unavoidable even with a shared archive and a less collision-prone
>> allocation scheme?
>>
>> I considered using some random increment of the TLI in the manner you describe but there are some issues that makes this
>> solution more complicated from an operational perspective:
>>
>> * If you skip some TLIs (in the sense pick a TLI that is "random but larger"), then it is not clear what the relation
>> between them are.
>>
>> * The history files contain the complete linkage of the timelines, so that is covered, but the naming would be strange.
>>
>> * For example, if you have history files 1, 5, 7, and 8, then these can all belong to different timelines, (except 1), or
>> be a single timeline and it is hard to understand which one without looking through the files.
>>
>> * With more promotions, the relation becomes even more strange, and the risk of collisions increases. (For example,
>> imagine one timeline with 1, 5, 7, 8, 11, and one timeline that forks off 1. Then any increment of 4, 6, 7, or 10 will
>> result in a collision.)
>>
>> * To actually reduce the risk significantly, you need to have a very wide range of the added randomness. Taking a smaller
>> number is easier to work with, but then you need to handle that some timelines can collide in some manner.
>> * Normally, the history file with the highest number will be the only relevant one. With this approach, you have to check
>> the contents of the files to understand which ones are relevant, which increases the operational burden.
>>
>> In contrast, if you use an UUID in this manner.
>>
>> * Adding an UUID does not require a central coordinator and is not likely to collide (on the level "impossible to
>> collide") and is very straightforward to add. It also comes with a low risk since the places in the code that requires
>> changes are very few and not likely to have unexpected consequences elsewhere. This works both with and without a
>> shared archive.
>> * Normally, a shared archive should only contain a single timeline. Anything else is an anomaly and should be corrected.
>> * I think it is still necessary to handle the case where you do not have a shared archive; it would be an odd limitation
>> to say that promote only works if you have a shared archive
>> * The UUID still serves a purpose in capturing a situation where things have gone wrong. Think of the UUID as similar to
>> a "checksum" safety and an extra precaution to prevent things from going wrong.
>>
>> In short, I think the operational issues with random increment of the history file number is worse, not better, and we
>> should deal with the name collisions correctly for shared archives instead. There is an issue in that it need to work
>> even in the case where you have a promotion that generates a new UUID but the correct history file exists (reported in
>> the other message) that I will look into.
>>
> I would like to know the current status of this patch. I have encountered the
> same issue in practice, and I think the proposed solution is reasonable.
>
> I found that the v6 patch does not apply cleanly to the current master (1f414035135)
> because commit 7f77b2a89bd4 changed the parameter type of writeTimeLineHistory().
>
> I've rebased the patch and attached v7.
>
>> Best wishes,
>> Mats Kindahl
>>
>> Best regards, Andrey Borodin.
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-08-30 17:35 Andrey Borodin <x4mmm@yandex-team.ru>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 1 reply; 35+ messages in thread
From: Andrey Borodin @ 2026-08-30 17:35 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: Japin Li <japinli@hotmail.com>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
Hi Mats,
On Tue, Aug 12, 2026, Mats Kindahl wrote:
> Even with a shared archive, it is impossible to ensure such things. It
> is a normal consensus problem.
This race came up again in an open-space discussion here. What do you
think about allowing a promotion request to supply the new TimelineID?
In managed HA setups, the tool that decides which node may promote
normally already has a DCS. It could allocate the TLI through the DCS
while selecting the new primary, then pass it to PostgreSQL. PostgreSQL
would still validate that it is greater than the current TLI and does
not conflict with any history file it can see. This would prevent the
collision when such a coordinator exists, while the UUID could remain a
last-resort check for uncoordinated promotions.
archive_command and restore_command are file-transfer interfaces. They
cannot express an atomic allocation, so they seem like the wrong place
to solve the distributed problem of choosing an identifier.
At least until PostgreSQL gets built-in Paxos. Every joke contains a
grain of joke, though; Kostya Osipov's related built-in consensus thread
is [0].
Would this fit the model you have in mind?
Thank you!
Best regards, Andrey Borodin.
[0] https://www.postgresql.org/message-id/flat/Z_1Cq7JvabsFYjQo%40ark
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-09-03 19:21 Mats Kindahl <mats.kindahl@gmail.com>
parent: Andrey Borodin <x4mmm@yandex-team.ru>
0 siblings, 0 replies; 35+ messages in thread
From: Mats Kindahl @ 2026-09-03 19:21 UTC (permalink / raw)
To: Andrey Borodin <x4mmm@yandex-team.ru>; +Cc: Japin Li <japinli@hotmail.com>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
Hi Andrey, and thank you for the comments.
On 8/30/26 19:35, Andrey Borodin wrote:
> Hi Mats,
>
> On Tue, Aug 12, 2026, Mats Kindahl wrote:
>> Even with a shared archive, it is impossible to ensure such things. It
>> is a normal consensus problem.
> This race came up again in an open-space discussion here. What do you
> think about allowing a promotion request to supply the new TimelineID?
> In managed HA setups, the tool that decides which node may promote
> normally already has a DCS. It could allocate the TLI through the DCS
> while selecting the new primary, then pass it to PostgreSQL. PostgreSQL
> would still validate that it is greater than the current TLI and does
> not conflict with any history file it can see. This would prevent the
> collision when such a coordinator exists, while the UUID could remain a
> last-resort check for uncoordinated promotions.
If we have some sort of API or callback for assigning a new TLI that
would be excellent. Would be useful for a bunch of DCS as a way to
ensure that you do not pick a conflicting TLI. Should probably be
configurable though, but a normal hook callback would probably be
sufficient... and yes, the UUID is good to have as a last-resort check
to ensure that you don't break things.
>
> archive_command and restore_command are file-transfer interfaces. They
> cannot express an atomic allocation, so they seem like the wrong place
> to solve the distributed problem of choosing an identifier.
Yup, agree.
>
> At least until PostgreSQL gets built-in Paxos. Every joke contains a
> grain of joke, though; Kostya Osipov's related built-in consensus thread
> is [0].
Would love to see that. Kostja would definitely know how to implement
that. :)
>
> Would this fit the model you have in mind?
That would work fine. My main concern about adding UUIDs were to prevent
problems when you accidentally allocate same TLI to timelines that are
not the same. Integration hooks with a DCS would be icing on the cake. :)
Best wishes,
Mats Kindahl
>
> Thank you!
>
>
> Best regards, Andrey Borodin.
>
> [0]https://www.postgresql.org/message-id/flat/Z_1Cq7JvabsFYjQo%40ark
>
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-09-19 11:51 Mats Kindahl <mats.kindahl@gmail.com>
parent: Japin Li <japinli@hotmail.com>
1 sibling, 1 reply; 35+ messages in thread
From: Mats Kindahl @ 2026-09-19 11:51 UTC (permalink / raw)
To: Japin Li <japinli@hotmail.com>; +Cc: Andrey Borodin <x4mmm@yandex-team.ru>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
Hi all,
I have created a commitfest issue
(https://commitfest.postgresql.org/patch/7317/) and also rebased the
patch on the latest HEAD (attached).
Best wishes,
Mats Kindahl, Multigres Engineer, Supabase
On 7/17/26 17:59, Japin Li wrote:
> Hi, all
>
> On Sun, 21 Jun 2026 at 11:09, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>> On 6/8/26 12:48, Andrey Borodin wrote:
>>
>> On 30 Apr 2026, at 13:19, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>>
>> There is one scenario that I assume is known that TLC found, but does not seem to be fixed. It is a relatively rare case, but since the fix is quite easy, I thought I'd share it with you and get feedback.
>>
>> Hi Mats,
>>
>> Hi Andrey,
>>
>> Thanks for looking at this.
>>
>> Thanks for working on this. I think the problem is real, but I wonder if
>> adding a separate UUID to timeline history files is solving it one step
>> too late.
>>
>> If two independent promotions manage to choose the same numeric TLI, then
>> we already have two different histories with the same timeline identifier.
>> Their history files will also have the same name. A UUID in the file lets
>> tools detect the mismatch afterwards, but it does not prevent the archive
>> namespace from containing two different meanings for the same TLI.
>>
>> Yes, that is correct.
>>
>> In normal deployments with a shared archive this should only be possible
>> when the history file is not visible to the other promoting server:
>> either there is no usable restore_command/shared archive, or there is a
>> race around publishing and observing the history file. In other words, TLI
>> allocation is not atomic, but it is intended to be coordinated through the
>> archive.
>>
>> Yes, that is the ideal way it should work when you have a shared archive. This works because you have a central authority
>> that synchronizes the timelines (in theory, not counting bugs).
>>
>> Maybe we should keep TimelineID as the actual branch identifier and make
>> that allocation harder to collide instead of adding a second identifier.
>> For example, when choosing a new TLI, add some randomness rather than just
>> using the next sequential value.
>>
>> That would make the race window much less
>> dangerous: two independent promotions would be extremely unlikely to
>> choose the same TLI, the history file names would remain distinct, and TLI
>> would keep its current role as the timeline identifier.
>> This also keeps the operational model simpler. TimelineID is already the
>> identifier exposed in WAL file names, history file names, logs, and
>> recovery configuration. If we add UUIDs, we effectively introduce another
>> identity for the same object, and tools then need to reason about both.
>> If instead we make TLI allocation less deterministic under races, the
>> existing model remains intact.
>>
>> Does that framing make sense, or am I missing a case where duplicate TLIs
>> are unavoidable even with a shared archive and a less collision-prone
>> allocation scheme?
>>
>> I considered using some random increment of the TLI in the manner you describe but there are some issues that makes this
>> solution more complicated from an operational perspective:
>>
>> * If you skip some TLIs (in the sense pick a TLI that is "random but larger"), then it is not clear what the relation
>> between them are.
>>
>> * The history files contain the complete linkage of the timelines, so that is covered, but the naming would be strange.
>>
>> * For example, if you have history files 1, 5, 7, and 8, then these can all belong to different timelines, (except 1), or
>> be a single timeline and it is hard to understand which one without looking through the files.
>>
>> * With more promotions, the relation becomes even more strange, and the risk of collisions increases. (For example,
>> imagine one timeline with 1, 5, 7, 8, 11, and one timeline that forks off 1. Then any increment of 4, 6, 7, or 10 will
>> result in a collision.)
>>
>> * To actually reduce the risk significantly, you need to have a very wide range of the added randomness. Taking a smaller
>> number is easier to work with, but then you need to handle that some timelines can collide in some manner.
>> * Normally, the history file with the highest number will be the only relevant one. With this approach, you have to check
>> the contents of the files to understand which ones are relevant, which increases the operational burden.
>>
>> In contrast, if you use an UUID in this manner.
>>
>> * Adding an UUID does not require a central coordinator and is not likely to collide (on the level "impossible to
>> collide") and is very straightforward to add. It also comes with a low risk since the places in the code that requires
>> changes are very few and not likely to have unexpected consequences elsewhere. This works both with and without a
>> shared archive.
>> * Normally, a shared archive should only contain a single timeline. Anything else is an anomaly and should be corrected.
>> * I think it is still necessary to handle the case where you do not have a shared archive; it would be an odd limitation
>> to say that promote only works if you have a shared archive
>> * The UUID still serves a purpose in capturing a situation where things have gone wrong. Think of the UUID as similar to
>> a "checksum" safety and an extra precaution to prevent things from going wrong.
>>
>> In short, I think the operational issues with random increment of the history file number is worse, not better, and we
>> should deal with the name collisions correctly for shared archives instead. There is an issue in that it need to work
>> even in the case where you have a promotion that generates a new UUID but the correct history file exists (reported in
>> the other message) that I will look into.
>>
> I would like to know the current status of this patch. I have encountered the
> same issue in practice, and I think the proposed solution is reasonable.
>
> I found that the v6 patch does not apply cleanly to the current master (1f414035135)
> because commit 7f77b2a89bd4 changed the parameter type of writeTimeLineHistory().
>
> I've rebased the patch and attached v7.
>
>> Best wishes,
>> Mats Kindahl
>>
>> Best regards, Andrey Borodin.
Attachments:
[text/x-patch] v9.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch (43.8K, ../../37d7b054-1b65-4fc9-87f9-839a1e791fc5@gmail.com/2-v9.0001-pg_rewind-use-UUIDs-to-detect-independent-same-TLI-p.patch)
download | inline diff:
From 0f7a2f8dd525d115fa47d84977fb2f51ef0a2905 Mon Sep 17 00:00:00 2001
From: Mats Kindahl <mats@kindahl.net>
Date: Sat, 23 May 2026 16:09:44 +0200
Subject: pg_rewind: use UUIDs to detect independent same-TLI promotions
Two PostgreSQL standbys can independently promote to the same timeline
ID if their primary stopped before either had a chance to promote. In
that situation both clusters share a timeline history prefix that looks
identical to pg_rewind: same TLI numbers and same begin/end LSNs. The
existing same-TLI shortcut therefore treated the source as a valid
rewind target and skipped the rewind entirely, leaving the target's
diverged WAL intact.
Fix this by embedding a UUIDv7 value in every timeline history file
entry at promotion time. Each promotion generates a fresh UUID, so two
independent promotions to the same TLI will carry different UUIDs even
though the TLI number and begin LSN are identical.
When loading the timeline history, pg_rewind uses these UUIDs in two
places:
1. findCommonAncestorTimeline checks that the TLI and UUID in each entry
match. A mismatch signals independent promotions and the search
continues to earlier entries to find the true common ancestor.
2. The same-TLI shortcut (source and target on the same current TLI)
compares the UUID stored in the last completed history entry and a
mismatch forces a full rewind instead of a no-op.
UUIDs are zero for clusters that predate this change, and the comparison
function treats a zero UUID on either side as different from a UUID
since that promotion has to be from a different server (it had a
pre-change version server that was promoted, so it cannot be the same as
a post-change version server that was promoted).
Before this commit, Postgres tolerated the target having no local copy
of the history file for its own current TLI. This could happen if, for
example, a standby that started streaming a timeline it never itself
switched to, hence had no reason to write or fetch a copy of that file.
This behaviour is retained.
Tests in t/005_same_timeline.pl cover these paths.
The first covers the same-TLI shortcut: two standbys independently
promote to TLI2 and TLI2', each with a distinct UUID.
The second covers the ancestor search: the target goes through TLI1 ->
TLI2 -> TLI3 while the source independently promoted so that it has a
timeline with TLI1 -> TLI2' -> TLI3'. The test ensures that
findCommonAncestorTimeline backs up to TLI1 as the true common ancestor
rather than accepting the numerically matching TLI2 entry.
The third covers the missing-history-file fallback: two standbys
independently promote to TLI2, the target's own TLI2 history file is
then removed, and pg_rewind is checked to no longer abort because of it.
---
src/backend/access/transam/timeline.c | 84 ++++-
src/backend/access/transam/xlog.c | 15 +
src/backend/utils/adt/uuid.c | 15 +-
src/bin/pg_rewind/pg_rewind.c | 159 +++++++-
src/bin/pg_rewind/t/005_same_timeline.pl | 459 +++++++++++++++++++++++
src/bin/pg_rewind/timeline.c | 58 ++-
src/include/access/timeline.h | 5 +-
src/include/access/xlog_internal.h | 1 +
src/include/utils/uuid.h | 10 +-
9 files changed, 772 insertions(+), 34 deletions(-)
diff --git a/src/backend/access/transam/timeline.c b/src/backend/access/transam/timeline.c
index d80c8ffe0a7..1d1175c1b34 100644
--- a/src/backend/access/transam/timeline.c
+++ b/src/backend/access/transam/timeline.c
@@ -42,6 +42,8 @@
#include "pgstat.h"
#include "storage/fd.h"
#include "utils/wait_event.h"
+#include "utils/fmgrprotos.h"
+#include "utils/uuid.h"
/*
* Copies all timeline history files with id's between 'begin' and 'end'
@@ -110,8 +112,12 @@ readTimeLineHistory(TimeLineID targetTLI)
ereport(FATAL,
(errcode_for_file_access(),
errmsg("could not open file \"%s\": %m", path)));
- /* Not there, so assume no parents */
- entry = palloc_object(TimeLineHistoryEntry);
+
+ /*
+ * Not there, so assume no parents. We use palloc0_object to ensure
+ * that tluuid is all-zero.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = entry->end = InvalidXLogRecPtr;
return list_make1(entry);
@@ -125,6 +131,7 @@ readTimeLineHistory(TimeLineID targetTLI)
prevend = InvalidXLogRecPtr;
for (;;)
{
+ char uuid_str[UUID_STR_LEN + 1] = {0};
char fline[MAXPGPATH];
char *res;
char *ptr;
@@ -155,7 +162,8 @@ readTimeLineHistory(TimeLineID targetTLI)
if (*ptr == '\0' || *ptr == '#')
continue;
- nfields = sscanf(fline, "%u\t%X/%08X", &tli, &switchpoint_hi, &switchpoint_lo);
+ nfields =
+ sscanf(fline, "%u\t%X/%08X\t%36s", &tli, &switchpoint_hi, &switchpoint_lo, uuid_str);
if (nfields < 1)
{
@@ -164,7 +172,7 @@ readTimeLineHistory(TimeLineID targetTLI)
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a numeric timeline ID.")));
}
- if (nfields != 3)
+ if (nfields < 3)
ereport(FATAL,
(errmsg("syntax error in history file: %s", fline),
errhint("Expected a write-ahead log switchpoint location.")));
@@ -176,12 +184,50 @@ readTimeLineHistory(TimeLineID targetTLI)
lasttli = tli;
- entry = palloc_object(TimeLineHistoryEntry);
+ /*
+ * We use palloc0_object to ensure that tluuid is all-zero, which is
+ * important for pg_rewind to detect whether the history file is
+ * missing or not.
+ */
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = tli;
entry->begin = prevend;
entry->end = ((uint64) (switchpoint_hi)) << 32 | (uint64) switchpoint_lo;
prevend = entry->end;
+ /*
+ * Parse the optional UUID field. Old history files have the reason
+ * string in field 4. It is in theory possible that the reason string
+ * starts with a UUID, but the current usage do not store a UUID. This
+ * allows us to support both old and new formats of history files
+ * without breaking compatibility by checking if the field contains a
+ * valid UUID.
+ */
+ if (nfields == 4 && strlen(uuid_str) == UUID_STR_LEN)
+ {
+ MemoryContext oldcontext = CurrentMemoryContext;
+
+ PG_TRY();
+ {
+ Datum datum = DirectFunctionCall1(uuid_in, CStringGetDatum(uuid_str));
+
+ memcpy(&entry->tluuid, DatumGetUUIDP(datum), sizeof(pg_uuid_t));
+ }
+ PG_CATCH();
+ {
+ ErrorData *edata;
+
+ MemoryContextSwitchTo(oldcontext);
+ edata = CopyErrorData();
+ FlushErrorState();
+
+ ereport(FATAL,
+ errmsg("invalid UUID in history file \"%s\"", path),
+ errdetail("%s", edata->message));
+ }
+ PG_END_TRY();
+ }
+
/* Build list with newest item first */
result = lcons(entry, result);
@@ -197,9 +243,11 @@ readTimeLineHistory(TimeLineID targetTLI)
/*
* Create one more entry for the "tip" of the timeline, which has no entry
- * in the history file.
+ * in the history file. We use palloc0_object to ensure that tluuid is
+ * all-zero, which is important for pg_rewind to detect whether the
+ * history file is missing or not.
*/
- entry = palloc_object(TimeLineHistoryEntry);
+ entry = palloc0_object(TimeLineHistoryEntry);
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
@@ -294,21 +342,33 @@ findNewestTimeLine(TimeLineID startTLI)
*
* newTLI: ID of the new timeline
* parentTLI: ID of its immediate parent
+ * newTLUUID: UUID uniquely identifying this promotion instance
* switchpoint: WAL location where the system switched to the new timeline
* reason: human-readable explanation of why the timeline was switched
*
- * Currently this is only used at the end recovery, and so there are no locking
+ * The output file is named <newTLI>.history (e.g. 00000003.history). If two
+ * servers independently promote to the same timeline ID, their history files
+ * share the same name. In a shared WAL archive the second file to arrive
+ * silently overwrites the first. The newTLUUID written into the file content
+ * lets pg_rewind detect this collision: it fetches each server's history file
+ * directly from that server, compares the UUIDs for every shared TLI, and
+ * treats a UUID mismatch as evidence of independent promotion even when the
+ * TLI numbers agree.
+ *
+ * Currently this is only used at end of recovery, and so there are no locking
* considerations. But we should be just as tense as XLogFileInit to avoid
* emplacing a bogus file.
*/
void
writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, const char *reason)
{
char path[MAXPGPATH];
char tmppath[MAXPGPATH];
char histfname[MAXFNAMELEN];
char buffer[BLCKSZ];
+ char *uuid_str;
int srcfd;
int fd;
ssize_t nbytes;
@@ -398,13 +458,19 @@ writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
*
* If we did have a parent file, insert an extra newline just in case the
* parent file failed to end with one.
+ *
+ * Format: <parentTLI>\t<switchpoint>\t<ThisTimeLineUUID>\t<reason>\n
*/
+ uuid_str = DatumGetCString(DirectFunctionCall1(uuid_out, UUIDPGetDatum(newTLUUID)));
+
snprintf(buffer, sizeof(buffer),
- "%s%u\t%X/%08X\t%s\n",
+ "%s%u\t%X/%08X\t%s\t%s\n",
(srcfd < 0) ? "" : "\n",
parentTLI,
LSN_FORMAT_ARGS(switchpoint),
+ uuid_str,
reason);
+ pfree(uuid_str);
nbytes = strlen(buffer);
errno = 0;
diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c
index 9ec0be77ca0..aa4e5add825 100644
--- a/src/backend/access/transam/xlog.c
+++ b/src/backend/access/transam/xlog.c
@@ -99,6 +99,7 @@
#include "storage/subsystems.h"
#include "storage/sync.h"
#include "utils/guc_hooks.h"
+#include "utils/uuid.h"
#include "utils/guc_tables.h"
#include "utils/injection_point.h"
#include "utils/pgstat_internal.h"
@@ -6644,6 +6645,9 @@ StartupXLOG(void)
newTLI = endOfRecoveryInfo->lastRecTLI;
if (ArchiveRecoveryRequested)
{
+ struct timeval tv;
+ pg_uuid_t uuid_buf;
+
newTLI = findNewestTimeLine(recoveryTargetTLI) + 1;
ereport(LOG,
(errmsg("selected new timeline ID: %u", newTLI)));
@@ -6674,8 +6678,19 @@ StartupXLOG(void)
* to the new timeline, and will try to connect to the new timeline.
* To minimize the window for that, try to do as little as possible
* between here and writing the end-of-recovery record.
+ *
+ * Generate a UUIDv7 that uniquely identifies this promotion. The
+ * same UUID is written into the history file so that pg_rewind can
+ * distinguish two servers that independently promoted to the same
+ * timeline ID. Use gettimeofday() since we are not on a hot path;
+ * generate_uuidv7 wants milliseconds and we pass 0 for sub-ms since
+ * the random bits already distinguish UUIDs generated within the same
+ * millisecond.
*/
+ gettimeofday(&tv, NULL);
+ generate_uuidv7_r(&uuid_buf, tv.tv_sec * 1000 + tv.tv_usec / 1000, 0);
writeTimeLineHistory(newTLI, recoveryTargetTLI,
+ &uuid_buf,
EndOfLog, endOfRecoveryInfo->recoveryStopReason);
ereport(LOG,
diff --git a/src/backend/utils/adt/uuid.c b/src/backend/utils/adt/uuid.c
index c019e17f21b..a6c010dca9b 100644
--- a/src/backend/utils/adt/uuid.c
+++ b/src/backend/utils/adt/uuid.c
@@ -91,7 +91,7 @@ static bool uuid_abbrev_abort(int memtupcount, SortSupport ssup);
static Datum uuid_abbrev_convert(Datum original, SortSupport ssup);
static inline void uuid_set_version(pg_uuid_t *uuid, unsigned char version);
static inline int64 get_real_time_ns_ascending(void);
-static pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
Datum
uuid_in(PG_FUNCTION_ARGS)
@@ -707,6 +707,14 @@ get_real_time_ns_ascending(void)
return ns;
}
+pg_uuid_t *
+generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+{
+ pg_uuid_t *uuid = palloc(UUID_LEN);
+
+ return generate_uuidv7_r(uuid, unix_ts_ms, sub_ms);
+}
+
/*
* Generate UUID version 7 per RFC 9562, with the given timestamp.
*
@@ -723,10 +731,9 @@ get_real_time_ns_ascending(void)
*
* NB: all numbers here are unsigned, unix_ts_ms cannot be negative per RFC.
*/
-static pg_uuid_t *
-generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms)
+pg_uuid_t *
+generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms)
{
- pg_uuid_t *uuid = palloc(UUID_LEN);
uint32 increased_clock_precision;
/* Fill in time part */
diff --git a/src/bin/pg_rewind/pg_rewind.c b/src/bin/pg_rewind/pg_rewind.c
index d2521dab333..3c087f55250 100644
--- a/src/bin/pg_rewind/pg_rewind.c
+++ b/src/bin/pg_rewind/pg_rewind.c
@@ -32,6 +32,19 @@
#include "rewind_source.h"
#include "storage/bufpage.h"
+/*
+ * Timeline histories for both clusters, populated by matchAndFetchTimelines().
+ */
+typedef struct TimeLineHistoriesData
+{
+ TimeLineHistoryEntry *source,
+ *target;
+ int sourceNentries,
+ targetNentries;
+} TimeLineHistoriesData;
+
+typedef TimeLineHistoriesData * TimeLineHistories;
+
static void usage(const char *progname);
static void perform_rewind(filemap_t *filemap, rewind_source *source,
@@ -54,6 +67,9 @@ static void findCommonAncestorTimeline(TimeLineHistoryEntry *a_history,
TimeLineHistoryEntry *b_history,
int b_nentries,
XLogRecPtr *recptr, int *tliIndex);
+static inline bool matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b);
+static bool matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli,
+ TimeLineHistories timelineHistories);
static void ensureCleanShutdown(const char *argv0);
static void disconnect_atexit(void);
@@ -142,6 +158,7 @@ main(int argc, char **argv)
int c;
XLogRecPtr divergerec;
int lastcommontliIndex;
+ TimeLineHistoriesData timelineHistories;
XLogRecPtr chkptrec;
TimeLineID chkpttli;
XLogRecPtr chkptredo;
@@ -376,10 +393,21 @@ main(int argc, char **argv)
*
* If both clusters are already on the same timeline, there's nothing to
* do.
+ *
+ * This also handles the case when two servers independently promoted to
+ * the same timeline ID: one crashed after writing the history file but
+ * before its EOR WAL record was distributed, so a second standby promoted
+ * independently. The history files produced by those two promotions
+ * carry different UUIDs.
+ *
+ * When the clusters are on different timelines we locate the fork point
+ * via findCommonAncestorTimeline.
*/
- if (target_tli == source_tli)
+ if (matchAndFetchTimelines(source_tli, target_tli, &timelineHistories))
{
pg_log_info("source and target cluster are on the same timeline");
+ pfree(timelineHistories.source);
+ pfree(timelineHistories.target);
rewind_needed = false;
target_wal_endrec = InvalidXLogRecPtr;
}
@@ -393,8 +421,10 @@ main(int argc, char **argv)
* Retrieve timelines for both source and target, and find the point
* where they diverged.
*/
- sourceHistory = getTimelineHistory(source_tli, true, &sourceNentries);
- targetHistory = getTimelineHistory(target_tli, false, &targetNentries);
+ targetHistory = timelineHistories.target;
+ targetNentries = timelineHistories.targetNentries;
+ sourceHistory = timelineHistories.source;
+ sourceNentries = timelineHistories.sourceNentries;
findCommonAncestorTimeline(sourceHistory, sourceNentries,
targetHistory, targetNentries,
@@ -982,40 +1012,69 @@ MinXLogRecPtr(XLogRecPtr a, XLogRecPtr b)
return Min(a, b);
}
+static bool
+file_exists(const char *name)
+{
+ struct stat st;
+
+ Assert(name != NULL);
+
+ if (stat(name, &st) == 0)
+ return !S_ISDIR(st.st_mode);
+ else if (!(errno == ENOENT || errno == ENOTDIR || errno == EACCES))
+ pg_fatal("could not stat file \"%s\": %m", name);
+
+ return false;
+
+}
+
/*
* Retrieve timeline history for the source or target system.
*/
static TimeLineHistoryEntry *
getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
{
- TimeLineHistoryEntry *history;
+ TimeLineHistoryEntry *history = NULL;
/*
* Timeline 1 does not have a history file, so there is no need to check
* and fake an entry with infinite start and end positions.
*/
- if (tli == 1)
- {
- history = pg_malloc_object(TimeLineHistoryEntry);
- history->tli = tli;
- history->begin = history->end = InvalidXLogRecPtr;
- *nentries = 1;
- }
- else
+ if (tli != 1)
{
char path[MAXPGPATH];
- char *histfile;
+ char *histfile = NULL;
TLHistoryFilePath(path, tli);
- /* Get history file from appropriate source */
+ /* Get history file from appropriate source, tolerating its absence */
if (is_source)
histfile = source->fetch_file(source, path, NULL);
else
- histfile = slurpFile(datadir_target, path, NULL);
+ {
+ char fullpath[MAXPGPATH];
+
+ snprintf(fullpath, sizeof(fullpath), "%s/%s", datadir_target, path);
+ if (file_exists(fullpath))
+ histfile = slurpFile(datadir_target, path, NULL);
+ }
+
+ if (histfile != NULL)
+ {
+ history = rewind_parseTimeLineHistory(histfile, tli, nentries);
+ pg_free(histfile);
+ }
+ }
- history = rewind_parseTimeLineHistory(histfile, tli, nentries);
- pg_free(histfile);
+ /*
+ * If no history file entry was created, we create a "zero" entry.
+ */
+ if (history == NULL)
+ {
+ history = pg_malloc0_object(TimeLineHistoryEntry);
+ history->tli = tli;
+ history->begin = history->end = InvalidXLogRecPtr;
+ *nentries = 1;
}
/* In debugging mode, print what we read */
@@ -1042,6 +1101,56 @@ getTimelineHistory(TimeLineID tli, bool is_source, int *nentries)
return history;
}
+/*
+ * Return true if two per-entry promotion UUIDs are compatible.
+ *
+ * A zero UUID means the history file predates this fix (or the entry is
+ * synthetic). If both sides are zero we have no UUID information and fall
+ * back to TLI-number-only matching (backward compatibility with old servers).
+ * If one side carries a UUID and the other does not, they cannot originate
+ * from the same promotion and are treated as incompatible.
+ */
+static inline bool
+matchingTimelineUUID(TimeLineHistoryEntry *a, TimeLineHistoryEntry *b)
+{
+ static const pg_uuid_t zero = {{0}};
+
+ if (memcmp(&a->tluuid, &zero, UUID_LEN) == 0 && memcmp(&b->tluuid, &zero, UUID_LEN) == 0)
+ return true;
+ return memcmp(&a->tluuid, &b->tluuid, UUID_LEN) == 0;
+}
+
+/*
+ * Fetch the timeline history for both clusters, store them in tlh, and return
+ * true if the clusters are on the same timeline (no rewind needed).
+ *
+ * tlh is always fully populated on return regardless of the result, so the
+ * caller can pass tlh->source / tlh->target directly to
+ * findCommonAncestorTimeline() when the return value is false.
+ *
+ * TLI 1 always returns true: it is the original timeline and has no promotion
+ * UUID. For TLI >= 2, the UUID in entry[Nentries - 2] identifies the
+ * promotion that created the current TLI. Both-zero UUIDs (old history files)
+ * are treated as compatible; zero-vs-nonzero is treated as a mismatch because
+ * one side carries a promotion UUID and they cannot be the same promotion.
+ */
+static bool
+matchAndFetchTimelines(TimeLineID source_tli, TimeLineID target_tli, TimeLineHistories tlh)
+{
+ tlh->source = getTimelineHistory(source_tli, true, &tlh->sourceNentries);
+ tlh->target = getTimelineHistory(target_tli, false, &tlh->targetNentries);
+
+ if (source_tli != target_tli)
+ return false;
+
+ /* TLI 1 has no promotion UUID; always treat as the same timeline. */
+ if (tlh->sourceNentries < 2 || tlh->targetNentries < 2)
+ return true;
+
+ return matchingTimelineUUID(&tlh->source[tlh->sourceNentries - 2],
+ &tlh->target[tlh->targetNentries - 2]);
+}
+
/*
* Determine the TLI of the last common timeline in the timeline history of
* two clusters. *tliIndex is set to the index of last common timeline in
@@ -1063,12 +1172,26 @@ findCommonAncestorTimeline(TimeLineHistoryEntry *a_history, int a_nentries,
* depending on the history files that each node has fetched in previous
* recovery processes. Hence check the start position of the new timeline
* as well and move down by one extra timeline entry if they do not match.
+ *
+ * We also compare timeline UUIDs when both sides carry one. Two servers
+ * that independently promoted to the same timeline ID produce history
+ * files with the same name (e.g. 00000003.history); in a shared WAL
+ * archive the second file silently overwrites the first. pg_rewind
+ * fetches each server's history file directly from that server, so it
+ * sees both UUIDs.
+ *
+ * The timeline UUID stored in history entry[i] is the UUID of the
+ * promotion that created entry[i+1], i.e. the UUID of TLI entry[i+1].tli.
+ * So to check whether entry[i] itself represents the same timeline on
+ * both sides we look at entry[i-1].tluuid (for i > 0). TLI 1 (i == 0) is
+ * always the same: it is the original timeline and has no promotion UUID.
*/
n = Min(a_nentries, b_nentries);
for (i = 0; i < n; i++)
{
if (a_history[i].tli != b_history[i].tli ||
- a_history[i].begin != b_history[i].begin)
+ a_history[i].begin != b_history[i].begin ||
+ (i > 0 && !matchingTimelineUUID(&a_history[i - 1], &b_history[i - 1])))
break;
}
diff --git a/src/bin/pg_rewind/t/005_same_timeline.pl b/src/bin/pg_rewind/t/005_same_timeline.pl
index 95a40c3b270..b1797f78881 100644
--- a/src/bin/pg_rewind/t/005_same_timeline.pl
+++ b/src/bin/pg_rewind/t/005_same_timeline.pl
@@ -7,6 +7,8 @@
#
use strict;
use warnings FATAL => 'all';
+use File::Copy;
+use PostgreSQL::Test::Cluster;
use PostgreSQL::Test::Utils;
use Test::More;
@@ -21,4 +23,461 @@ RewindTest::create_standby();
RewindTest::run_pg_rewind('local');
RewindTest::clean_rewind_test();
+# Helper function to run pg_rewind in local mode with the given source and
+# target nodes and extra arguments.
+#
+# The target and source nodes are stopped before the call and the target is
+# restarted afterward. The target's postgresql.conf is copied to a temporary
+# location and passed to pg_rewind with --config-file, so that pg_rewind can
+# update the target's config file in place without worrying about file
+# permissions. The temporary config file is moved back to the target's data
+# directory and permissions fixed after pg_rewind finishes.
+#
+# If the last element of @extra_args is a coderef, it is not passed to
+# pg_rewind: it is run after pg_rewind finishes but before the target is
+# restarted, e.g. to undo a test's own tampering with the target's data
+# directory that pg_rewind itself doesn't need to be aware of.
+sub rewind_node
+{
+ my ($target, $source, $label, %opts) = @_;
+ my @extra_args;
+
+ $source->stop;
+ $target->stop;
+
+ push @extra_args, '--restore-target-wal' if $opts{'-restorewal'};
+
+ my $tpgdata = $target->data_dir;
+ my $tmp = PostgreSQL::Test::Utils::tempdir;
+ copy("$tpgdata/postgresql.conf", "$tmp/target-postgresql.conf.tmp");
+
+ command_ok(
+ [
+ 'pg_rewind',
+ '--debug',
+ '--source-pgdata' => $source->data_dir,
+ '--target-pgdata' => $target->data_dir,
+ '--no-sync',
+ '--config-file' => "$tmp/target-postgresql.conf.tmp",
+ @extra_args,
+ ],
+ $label);
+
+ move("$tmp/target-postgresql.conf.tmp", "$tpgdata/postgresql.conf");
+ chmod($target->group_access() ? 0640 : 0600, "$tpgdata/postgresql.conf")
+ or BAIL_OUT("unable to set permissions for $tpgdata/postgresql.conf");
+
+ $target->start unless $opts{'-norestart'};
+}
+
+# Rewrite a node's TLI history file in the old 3-field format (no UUID), so
+# that pg_rewind sees a zero UUID for that side, as if the node had been
+# promoted by a server that predates the UUID feature.
+sub strip_tli_uuid
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ my @lines = <$fh>;
+ close $fh;
+ open($fh, '>', $histfile) or die "cannot write $histfile: $!";
+ for my $line (@lines)
+ {
+ chomp $line;
+ my @f = split(/\t/, $line, 4);
+ if (@f == 4)
+ {
+
+ # Drop the UUID field (index 2); keep parentTLI, switchpoint, reason.
+ print $fh join("\t", $f[0], $f[1], $f[3]) . "\n";
+ }
+ else
+ {
+ print $fh "$line\n";
+ }
+ }
+ close $fh;
+}
+
+# Remove a node's TLI history file entirely, simulating a node that has no local
+# record of its own current-TLI promotion, for example, because it started
+# streaming a timeline it never itself switched to. This means that it never had
+# a reason to write (or fetch) a local copy of that history file. Returns the
+# removed file's content, so it can be restored with restore_tli_history().
+#
+# Note this simulation is imperfect: unlike a node that never had that reason,
+# our node genuinely did promote and so still has the earlier TLI's real WAL
+# segments on disk. Without the history file linking them, the server itself can
+# no longer make sense of its own pg_wal directory and will fail to start --
+# restore_tli_history() must be called before restarting the node.
+sub remove_tli_history
+{
+ my ($node, $tli) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '<', $histfile) or die "cannot open $histfile: $!";
+ local $/;
+ my $content = <$fh>;
+ close $fh;
+ unlink($histfile) or die "cannot remove $histfile: $!";
+ return $content;
+}
+
+# Restore a TLI history file previously removed by remove_tli_history().
+sub restore_tli_history
+{
+ my ($node, $tli, $content) = @_;
+ my $histfile = sprintf("%s/pg_wal/%08X.history", $node->data_dir, $tli);
+ open(my $fh, '>', $histfile) or die "cannot write $histfile: $!";
+ print $fh $content;
+ close $fh;
+}
+
+# Helper function to create an origin node with a test table and a row containing
+# the given label. The node is started and ready for use as a source for
+# standbys.
+sub setup_origin
+{
+ my ($label) = @_;
+ my $node = PostgreSQL::Test::Cluster->new($label);
+ $node->init(allows_streaming => 1);
+ $node->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $node->start;
+ $node->safe_psql('postgres', "CREATE TABLE tbl (val text)");
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+ return $node;
+}
+
+# Helper function to create multiple standby nodes from the same origin node.
+# Each standby gets its own backup and data directory, so that they will
+# generate independent UUIDs on promotion even though they share the same
+# timeline history up to the point of promotion.
+sub setup_standbys_from_origin
+{
+ my ($origin, @names) = @_;
+ my @standbys;
+ for my $name (@names)
+ {
+ my $standby = PostgreSQL::Test::Cluster->new($name);
+ $origin->backup($standby->name);
+ $standby->init_from_backup($origin, $standby->name,
+ has_streaming => 1);
+ $standby->append_conf('postgresql.conf', "wal_keep_size = 320MB\n");
+ $standby->set_standby_mode();
+ $standby->start;
+ push @standbys, $standby;
+ }
+ return @standbys;
+}
+
+# Helper function to wait for multiple standby nodes to catch up to the origin.
+sub sync_standbys_with_origin
+{
+ my ($origin, @standbys) = @_;
+ $origin->wait_for_catchup($_) for @standbys;
+}
+
+# Helper function to insert a row with the given label into a node's test table.
+sub write_record
+{
+ my ($node, $label) = @_;
+ $node->safe_psql('postgres', "INSERT INTO tbl VALUES ('$label')");
+ $node->safe_psql('postgres', 'CHECKPOINT');
+}
+
+# Test that pg_rewind detects and handles two standbys that independently
+# promoted to the same timeline ID. Before the UUID-based divergence check,
+# pg_rewind's same-TLI shortcut would incorrectly skip the rewind in this
+# case, leaving the target's diverged WAL intact.
+#
+# origin (TLI 1)
+# |
+# +--- node_a (TLI 1) --promote--> TLI 2, UUID-A (target)
+# |
+# +--- node_b (TLI 1) --promote--> TLI 2, UUID-B (source)
+#
+# pg_rewind must detect the UUID mismatch and rewind node_a to match node_b.
+
+my $node_origin = setup_origin('origin');
+
+# Create node_a and node_b from separate backups of origin so that each
+# has its own data directory and will generate an independent UUID on promotion.
+my ($node_a, $node_b) =
+ setup_standbys_from_origin($node_origin, 'node_a', 'node_b');
+
+# Wait for both standbys to catch up to origin, then stop origin. After
+# this point the two standbys are isolated and will promote independently.
+sync_standbys_with_origin($node_origin, $node_a, $node_b);
+$node_origin->stop;
+
+# Promote both standbys. Each lands on TLI 2 but generates a distinct UUID,
+# so the resulting clusters are diverged even though they share a timeline ID.
+$node_a->promote;
+$node_b->promote;
+
+# Insert a divergent row on each so the rewind has visible work to do.
+write_record($node_a, 'in A');
+write_record($node_b, 'in B');
+
+rewind_node($node_a, $node_b,
+ 'pg_rewind detects independent same-TLI promotions');
+
+my $result =
+ $node_a->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result, "in B\norigin",
+ 'rewound node has source data, not its own divergent data');
+
+$node_a->teardown_node;
+$node_b->teardown_node;
+$node_origin->teardown_node;
+
+# Test that pg_rewind correctly rewinds across a TLI mismatch buried in a shared
+# prefix of the timeline history. The target has gone through three timelines
+# (TLI 1 -> TLI 2 -> TLI 3) while the source independently promoted from TLI 1
+# to what is numerically TLI 2 but with a different UUID (TLI 2'). The deepest
+# common ancestor is therefore TLI 1, and pg_rewind must rewind the target all
+# the way back to the end of TLI 1.
+#
+# origin (TLI 1) --+-- node_x --promote--> TLI 2 -- node_a --promote--> TLI 3
+# | (target: TLI 1->TLI 2->TLI 3)
+# +-- node_b --promote--> TLI 2'
+# (source: TLI 1->TLI 2')
+#
+# findCommonAncestorTimeline walks forward: TLI 1 entries match (UUID=0 on
+# both sides), then TLI 2 vs TLI 2' match on tli and begin but differ on
+# UUID, signalling independent promotions. The algorithm therefore backs up
+# to TLI 1 as the common ancestor and sets the divergence point to the end
+# of TLI 1.
+
+my $node_origin2 = setup_origin('origin2');
+
+# node_x and node_b2 both start from the same TLI 1 baseline.
+my ($node_x, $node_b2) =
+ setup_standbys_from_origin($node_origin2, 'node_x', 'node_b2');
+
+# Both standbys must be caught up to the same LSN before origin stops, so
+# that TLI 2 and TLI 2' both begin at the same WAL position.
+sync_standbys_with_origin($node_origin2, $node_x, $node_b2);
+$node_origin2->stop;
+
+# Promote node_x to TLI 2 (UUID-X) and insert a row. node_b2 is still on
+# TLI 1 and has not yet seen any TLI 2 WAL.
+$node_x->promote;
+write_record($node_x, 'x');
+
+# Build node_a2 as a standby of node_x, then promote it to TLI 3.
+my ($node_a2) = setup_standbys_from_origin($node_x, 'node_a2');
+
+sync_standbys_with_origin($node_x, $node_a2);
+$node_x->stop;
+
+$node_a2->promote;
+
+# Now promote node_b2 independently from TLI 1 to TLI 2' (UUID-B, != UUID-X).
+$node_b2->promote;
+write_record($node_b2, 'b');
+
+# Rewind node_a2 (TLI 1->TLI 2->TLI 3) from node_b2 (TLI 1->TLI 2') in
+# local mode. The rewind must reach back to the end of TLI 1.
+#
+# node_a2 was initialised from a streaming backup of node_x taken after
+# node_x had already completed segment 4 of TLI 2; that segment therefore
+# does not appear in node_a2's pg_wal. pg_rewind's backward scan for the
+# last checkpoint before the divergence point needs that segment, so we
+# point restore_command at node_x's pg_wal and use --restore-target-wal.
+#
+# Note: no row is inserted on TLI 3. This is intentional: the only
+# post-divergence table modification in the target's WAL is the 'x' INSERT
+# on TLI 2. On unpatched code the WAL scan would start from the TLI 2
+# shutdown checkpoint (just before TLI 3), miss that earlier insert, and
+# leave 'x' in place instead of replacing it with 'b'.
+my $node_x_waldir = $node_x->data_dir . "/pg_wal";
+if ($PostgreSQL::Test::Utils::windows_os)
+{
+ $node_x_waldir =~ s{\\}{\\\\}g;
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'copy "$node_x_waldir\\\\%f" "%p"'\n));
+}
+else
+{
+ $node_a2->append_conf('postgresql.conf',
+ qq(\nrestore_command = 'cp "$node_x_waldir/%f" "%p"'\n));
+}
+
+rewind_node(
+ $node_a2, $node_b2,
+ 'pg_rewind rewinds across mismatched TLI 2 / TLI 2-prime to TLI 1',
+ -restorewal => 1);
+my $result2 =
+ $node_a2->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result2, "b\norigin2",
+ 'rewound node reflects source history, not target TLI 2/TLI 3 data');
+
+$node_a2->teardown_node;
+$node_b2->teardown_node;
+$node_x->teardown_node;
+$node_origin2->teardown_node;
+
+# Test that pg_rewind correctly detects a mismatch when one cluster's TLI 2
+# history entry carries a zero UUID (old-format history file) while the other
+# carries a real UUID. The two clusters must have promoted independently, so
+# pg_rewind must rewind to TLI 1 rather than accepting the same-TLI shortcut.
+#
+# Run both orientations:
+# (a) target has zero UUID, source has real UUID
+# (b) target has real UUID, source has zero UUID
+#
+# In both cases the setup is:
+#
+# origin (TLI 1) --+-- node_p --promote--> TLI 2, UUID-P (target)
+# |
+# +-- node_q --promote--> TLI 2, UUID-Q (source)
+#
+# One side then has its history file rewritten to the old 3-field format so
+# that its UUID reads as zero. pg_rewind must treat zero-vs-nonzero as
+# incompatible (they cannot be the same promotion) and rewind to TLI 1.
+
+for my $strip_target (1, 0)
+{
+ my $zero_side = $strip_target ? 'target' : 'source';
+ my $real_side = $strip_target ? 'source' : 'target';
+ my $sfx = $strip_target ? 'zt' : 'zs';
+ my $label =
+ "pg_rewind rewinds when $zero_side has zero UUID and $real_side has real UUID";
+
+ my $node_origin3 = setup_origin("origin3_$sfx");
+ my ($node_p, $node_q) =
+ setup_standbys_from_origin($node_origin3, "node_p_$sfx", "node_q_$sfx");
+
+ sync_standbys_with_origin($node_origin3, $node_p, $node_q);
+ $node_origin3->stop;
+
+ $node_p->promote;
+ $node_q->promote;
+
+ write_record($node_p, 'in P');
+ write_record($node_q, 'in Q');
+
+ # Strip UUID from the chosen side to simulate a pre-UUID server.
+ strip_tli_uuid($strip_target ? $node_p : $node_q, 2);
+
+ rewind_node($node_p, $node_q, $label);
+ my $result3 =
+ $node_p->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+ is( $result3,
+ "in Q\norigin3_$sfx",
+ 'rewound node has source data, not its own divergent row');
+
+ $node_p->teardown_node;
+ $node_q->teardown_node;
+ $node_origin3->teardown_node;
+}
+
+# Test that pg_rewind detects independent promotions to TLI 3 when both
+# clusters share a common TLI 1 -> TLI 2 history (same UUID) but independently
+# promoted from TLI 2 to TLI 3, producing different TLI 3 UUIDs.
+#
+# origin (TLI 1) --- node_mid --promote--> TLI 2, UUID-M
+# |
+# +-- node_c --promote--> TLI 3, UUID-C (target)
+# |
+# +-- node_d --promote--> TLI 3', UUID-D (source)
+#
+# The same-TLI shortcut compares entry[Nentries-2].tluuid on each side; that
+# is the UUID of the TLI 3 promotion, which differs. The full rewind path
+# then walks the history forward: TLI 1 matches (same tli/begin/UUID-M at
+# entry[0]), TLI 2 also matches (same tli/begin; UUID-M is the same on both
+# sides at entry[0]), but TLI 3 vs TLI 3' differ at entry[1] (UUID-C != UUID-D),
+# so the divergence point is set to the end of TLI 2.
+
+my $node_origin4 = setup_origin('origin4');
+my ($node_mid) = setup_standbys_from_origin($node_origin4, 'node_mid');
+
+sync_standbys_with_origin($node_origin4, $node_mid);
+$node_origin4->stop;
+
+# Promote node_mid to TLI 2 and insert a row that both TLI 3 nodes will share.
+$node_mid->promote;
+write_record($node_mid, 'mid');
+
+# node_c and node_d both start as standbys of node_mid so they share the same
+# TLI 2 promotion UUID (UUID-M).
+my ($node_c, $node_d) =
+ setup_standbys_from_origin($node_mid, 'node_c', 'node_d');
+sync_standbys_with_origin($node_mid, $node_c, $node_d);
+$node_mid->stop;
+
+# Promote both independently; each generates a distinct TLI 3 UUID.
+$node_c->promote;
+$node_d->promote;
+
+write_record($node_c, 'c');
+write_record($node_d, 'd');
+
+rewind_node($node_c, $node_d,
+ 'pg_rewind detects independent TLI 3 / TLI 3-prime promotions sharing TLI 2'
+);
+my $result4 =
+ $node_c->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is($result4, "d\nmid\norigin4",
+ 'rewound node has source TLI 3-prime data, not its own TLI 3 data');
+
+$node_c->teardown_node;
+$node_d->teardown_node;
+$node_mid->teardown_node;
+$node_origin4->teardown_node;
+
+# Test that pg_rewind does not fail when the target has no local copy of its own
+# current TLI's history file, for example, because it started streaming that TLI
+# from a primary that had already promoted, without ever itself performing a
+# promotion or otherwise obtaining a local copy of the file.
+#
+# Without that file, pg_rewind cannot read the promotion UUID for the shared TLI
+# on the target side, and so cannot tell whether the two clusters are the same
+# promotion or two independent ones.
+#
+# We check that no rewind does actually take place, not just that the error
+# message is correct. For that reason, we create two divergent nodes for the
+# test, remove the history file, and then attempt a rewind. If the code works
+# correctly, the node should not be rewound.
+my $node_origin5 = setup_origin('origin5');
+my ($node_g, $node_h) =
+ setup_standbys_from_origin($node_origin5, 'node_g', 'node_h');
+
+sync_standbys_with_origin($node_origin5, $node_g, $node_h);
+$node_origin5->stop;
+
+$node_g->promote;
+$node_h->promote;
+
+write_record($node_g, 'in G');
+write_record($node_h, 'in H');
+
+my $saved_history = remove_tli_history($node_g, 2);
+
+rewind_node(
+ $node_g, $node_h,
+ 'pg_rewind does not fail when target history file is missing',
+ -norestart => 1);
+
+# Restore the history file of node_g before it gets restarted. We are testing
+# that pg_rewind can handle a missing history file, but unlike the real-world
+# case this simulates, the pg_wal directory of node_g still has its own genuine
+# TLI 1 segments on disk and the server cannot start back up without the file
+# linking them to TLI 2.
+restore_tli_history($node_g, 2, $saved_history);
+
+$node_g->restart;
+
+my $result5 =
+ $node_g->safe_psql('postgres', "SELECT val FROM tbl ORDER BY val");
+is( $result5,
+ "in G\norigin5",
+ 'missing history file prevents divergence detection: target data left unchanged'
+);
+
+$node_g->teardown_node;
+$node_h->teardown_node;
+$node_origin5->teardown_node;
+
done_testing();
diff --git a/src/bin/pg_rewind/timeline.c b/src/bin/pg_rewind/timeline.c
index f4e1d7f7d1b..6062d78e30b 100644
--- a/src/bin/pg_rewind/timeline.c
+++ b/src/bin/pg_rewind/timeline.c
@@ -9,10 +9,41 @@
*/
#include "postgres_fe.h"
+#include <ctype.h>
+#include <string.h>
+
#include "access/timeline.h"
#include "common/pg_parse_lsn.h"
#include "pg_rewind.h"
+/*
+ * Parse a UUID string in standard dashed form into a pg_uuid_t.
+ * Returns true on success, false if str is not a valid UUID string.
+ */
+static bool
+rewind_parse_uuid(const char *str, pg_uuid_t *uuid)
+{
+ const char *src = str;
+
+ for (int i = 0; i < UUID_LEN; i++)
+ {
+ char buf[3];
+
+ if (!isxdigit((unsigned char) src[0]) ||
+ !isxdigit((unsigned char) src[1]))
+ return false;
+ buf[0] = src[0];
+ buf[1] = src[1];
+ buf[2] = '\0';
+ uuid->data[i] = (unsigned char) strtoul(buf, NULL, 16);
+ src += 2;
+ /* skip dash at positions after bytes 3, 5, 7, 9 (i == 3,5,7,9) */
+ if (src[0] == '-' && (i == 3 || i == 5 || i == 7 || i == 9))
+ src++;
+ }
+ return (*src == '\0');
+}
+
/*
* This is copy-pasted from the backend readTimeLineHistory, modified to
* return a malloc'd array and to work without backend functions.
@@ -52,6 +83,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
bool valid_switchpoint = false;
int nchars;
size_t nspaces;
+ char uuid_str[UUID_STR_LEN + 1] = {0};
fline = bufptr;
while (*bufptr && *bufptr != '\n')
@@ -94,6 +126,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
*token_end = '\0';
valid_switchpoint = pg_parse_lsn(ptr, &switchpoint);
*token_end = save;
+ ptr = token_end;
}
if (!valid_switchpoint)
@@ -120,7 +153,29 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->end = switchpoint;
prevend = entry->end;
- /* we ignore the remainder of each line */
+ /*
+ * Parse the optional UUID field that follows the switchpoint. Old
+ * history files have a human-readable reason string there instead;
+ * its first word is much shorter than UUID_STR_LEN, so the length
+ * check safely distinguishes old from new format. We ignore the
+ * remainder of the line beyond the UUID field.
+ */
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
+ nspaces = strspn(ptr, " \t\n\r\f\v");
+ if (nspaces > 0)
+ {
+ pg_uuid_t buf;
+
+ ptr += nspaces;
+ token_end = ptr + strcspn(ptr, " \t\n\r\f\v");
+ if ((size_t) (token_end - ptr) == UUID_STR_LEN)
+ {
+ memcpy(uuid_str, ptr, UUID_STR_LEN);
+ uuid_str[UUID_STR_LEN] = '\0';
+ if (rewind_parse_uuid(uuid_str, &buf))
+ memcpy(&entry->tluuid, &buf, sizeof(pg_uuid_t));
+ }
+ }
}
if (entries && targetTLI <= lasttli)
@@ -144,6 +199,7 @@ rewind_parseTimeLineHistory(char *buffer, TimeLineID targetTLI, int *nentries)
entry->tli = targetTLI;
entry->begin = prevend;
entry->end = InvalidXLogRecPtr;
+ memset(&entry->tluuid, 0, sizeof(pg_uuid_t));
*nentries = nlines;
return entries;
diff --git a/src/include/access/timeline.h b/src/include/access/timeline.h
index 3aee3419a5c..3b94c4f0a4d 100644
--- a/src/include/access/timeline.h
+++ b/src/include/access/timeline.h
@@ -13,6 +13,7 @@
#include "access/xlogdefs.h"
#include "nodes/pg_list.h"
+#include "utils/uuid.h"
/*
* A list of these structs describes the timeline history of the server. Each
@@ -22,9 +23,10 @@
* pointers of all the entries form a contiguous line from beginning of time
* to infinity.
*/
-typedef struct
+typedef struct TimeLineHistoryEntry
{
TimeLineID tli;
+ pg_uuid_t tluuid; /* from history file; zero if unknown */
XLogRecPtr begin; /* inclusive */
XLogRecPtr end; /* exclusive, InvalidXLogRecPtr means infinity */
} TimeLineHistoryEntry;
@@ -33,6 +35,7 @@ extern List *readTimeLineHistory(TimeLineID targetTLI);
extern bool existsTimeLineHistory(TimeLineID probeTLI);
extern TimeLineID findNewestTimeLine(TimeLineID startTLI);
extern void writeTimeLineHistory(TimeLineID newTLI, TimeLineID parentTLI,
+ const pg_uuid_t *newTLUUID,
XLogRecPtr switchpoint, const char *reason);
extern void writeTimeLineHistoryFile(TimeLineID tli, const char *content, size_t size);
extern void restoreTimeLineHistoryFiles(TimeLineID begin, TimeLineID end);
diff --git a/src/include/access/xlog_internal.h b/src/include/access/xlog_internal.h
index bf609c3c703..cef9a5ddee9 100644
--- a/src/include/access/xlog_internal.h
+++ b/src/include/access/xlog_internal.h
@@ -22,6 +22,7 @@
#include "access/xlogdefs.h"
#include "access/xlogreader.h"
#include "datatype/timestamp.h"
+#include "utils/uuid.h"
#include "lib/stringinfo.h"
#include "pgtime.h"
#include "storage/block.h"
diff --git a/src/include/utils/uuid.h b/src/include/utils/uuid.h
index 572d8cf4c36..47a6af0ab3c 100644
--- a/src/include/utils/uuid.h
+++ b/src/include/utils/uuid.h
@@ -17,12 +17,16 @@
/* uuid size in bytes */
#define UUID_LEN 16
+/* length of a UUID string (without null terminator): xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx */
+#define UUID_STR_LEN 36
+
typedef struct pg_uuid_t
{
unsigned char data[UUID_LEN];
} pg_uuid_t;
-/* fmgr interface macros */
+/* fmgr interface macros (backend only) */
+#ifndef FRONTEND
static inline Datum
UUIDPGetDatum(const pg_uuid_t *X)
{
@@ -39,4 +43,8 @@ DatumGetUUIDP(Datum X)
#define PG_GETARG_UUID_P(X) DatumGetUUIDP(PG_GETARG_DATUM(X))
+extern pg_uuid_t *generate_uuidv7(uint64 unix_ts_ms, uint32 sub_ms);
+extern pg_uuid_t *generate_uuidv7_r(pg_uuid_t *uuid, uint64 unix_ts_ms, uint32 sub_ms);
+
+#endif /* !FRONTEND */
#endif /* UUID_H */
--
2.43.0
^ permalink raw reply [nested|flat] 35+ messages in thread
* Re: pg_rewind does not rewind diverging timelines
@ 2026-09-27 17:16 Andrey Borodin <x4mmm@yandex-team.ru>
parent: Mats Kindahl <mats.kindahl@gmail.com>
0 siblings, 0 replies; 35+ messages in thread
From: Andrey Borodin @ 2026-09-27 17:16 UTC (permalink / raw)
To: Mats Kindahl <mats.kindahl@gmail.com>; +Cc: Japin Li <japinli@hotmail.com>; pgsql-hackers mailing list <pgsql-hackers@lists.postgresql.org>
> On 19 Sep 2026, at 16:51, Mats Kindahl <mats.kindahl@gmail.com> wrote:
>
> Hi all,
>
> I have created a commitfest issue (https://commitfest.postgresql.org/patch/7317/) and also rebased the patch on the latest HEAD (attached).
It's also registered as [0]. I propose to keep one just CF entry and close other
as duplicate. Perhaps, it's slightly better to close new one - Japin is registered
as reviewer in old one, at that would be correct to keep this records.
Thank you!
Best regards, Andrey Borodin.
[0] https://commitfest.postgresql.org/patch/6768/
^ permalink raw reply [nested|flat] 35+ messages in thread
end of thread, other threads:[~2026-09-27 17:16 UTC | newest]
Thread overview: 35+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2025-10-31 13:07 [PATCH 3/3] Initial steps to making relkind an enum Álvaro Herrera <alvherre@kurilemu.de>
2026-04-30 08:19 pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-05-01 16:06 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-05-21 22:09 ` Re: pg_rewind does not rewind diverging timelines surya poondla <suryapoondla4@gmail.com>
2026-05-24 18:30 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-05-25 05:20 ` Re: pg_rewind does not rewind diverging timelines Japin Li <japinli@hotmail.com>
2026-05-25 18:59 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-05-26 06:56 ` Re: pg_rewind does not rewind diverging timelines Japin Li <japinli@hotmail.com>
2026-05-26 16:03 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-05-29 02:01 ` Re: pg_rewind does not rewind diverging timelines Japin Li <japinli@hotmail.com>
2026-05-30 20:26 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-06-01 02:59 ` Re: pg_rewind does not rewind diverging timelines Japin Li <japinli@hotmail.com>
2026-06-01 06:30 ` Re: pg_rewind does not rewind diverging timelines Kyotaro Horiguchi <horikyota.ntt@gmail.com>
2026-06-01 20:32 ` Re: pg_rewind does not rewind diverging timelines surya poondla <suryapoondla4@gmail.com>
2026-06-02 02:13 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-06-02 05:29 ` Re: pg_rewind does not rewind diverging timelines Kyotaro Horiguchi <horikyota.ntt@gmail.com>
2026-07-26 23:36 ` Re: pg_rewind does not rewind diverging timelines Michael Paquier <michael@paquier.xyz>
2026-07-29 11:51 ` Re: pg_rewind does not rewind diverging timelines Ants Aasma <ants.aasma@cybertec.at>
2026-08-12 06:32 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-07-29 12:55 ` Re: pg_rewind does not rewind diverging timelines Andreas Karlsson <andreas@proxel.se>
2026-08-12 06:33 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-08-12 06:14 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-08-12 05:05 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-06-08 19:52 ` Re: pg_rewind does not rewind diverging timelines Zsolt Parragi <zsolt.parragi@percona.com>
2026-07-26 05:11 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-07-26 09:57 ` Re: pg_rewind does not rewind diverging timelines Tatsuya Kawata <kawatatatsuya0913@gmail.com>
2026-08-16 11:37 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-06-08 10:48 ` Re: pg_rewind does not rewind diverging timelines Andrey Borodin <x4mmm@yandex-team.ru>
2026-06-21 09:09 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-07-17 15:59 ` Re: pg_rewind does not rewind diverging timelines Japin Li <japinli@hotmail.com>
2026-08-16 11:54 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-08-30 17:35 ` Re: pg_rewind does not rewind diverging timelines Andrey Borodin <x4mmm@yandex-team.ru>
2026-09-03 19:21 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-09-19 11:51 ` Re: pg_rewind does not rewind diverging timelines Mats Kindahl <mats.kindahl@gmail.com>
2026-09-27 17:16 ` Re: pg_rewind does not rewind diverging timelines Andrey Borodin <x4mmm@yandex-team.ru>
This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox