agora inbox for [email protected]  
help / color / mirror / Atom feed
[PATCH 8/8] batch build
57+ messages / 10 participants
[nested] [flat]

* [PATCH 8/8] batch build
@ 2021-01-22 00:11  Tomas Vondra <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Tomas Vondra @ 2021-01-22 00:11 UTC (permalink / raw)

---
 src/backend/access/brin/brin_minmax_multi.c | 204 ++++++++++++++++++--
 src/backend/access/brin/brin_tuple.c        |   1 +
 src/include/access/brin_tuple.h             |   2 +-
 3 files changed, 186 insertions(+), 21 deletions(-)

diff --git a/src/backend/access/brin/brin_minmax_multi.c b/src/backend/access/brin/brin_minmax_multi.c
index 094ecd2be9..72cca6f26b 100644
--- a/src/backend/access/brin/brin_minmax_multi.c
+++ b/src/backend/access/brin/brin_minmax_multi.c
@@ -57,6 +57,7 @@
 #include "access/brin.h"
 #include "access/brin_internal.h"
 #include "access/brin_tuple.h"
+#include "access/hash.h"	/* XXX strange that it fails because of BRIN_AM_OID without this */
 #include "access/reloptions.h"
 #include "access/stratnum.h"
 #include "access/htup_details.h"
@@ -150,12 +151,23 @@ typedef struct MinMaxOptions
 typedef struct Ranges
 {
 	Oid		typid;
+	Oid		colloid;
+	AttrNumber	attno;
 
 	/* (2*nranges + nvalues) <= maxvalues */
 	int		nranges;	/* number of ranges in the array (stored) */
 	int		nvalues;	/* number of values in the data array (all) */
 	int		maxvalues;	/* maximum number of values (reloption) */
 
+	/*
+	 * Batch mode means we're simply appending values into new range,
+	 * without any expensive steps (sorting, deduplication, ...). The
+	 * buffer is sized to be much larger, but we keep the actual target
+	 * size so that when serializing values we can do the right thing.
+	 */
+	bool	batch_mode;
+	int		target_maxvalues;
+
 	/* values stored for this range - either raw values, or ranges */
 	Datum	values[FLEXIBLE_ARRAY_MEMBER];
 } Ranges;
@@ -375,6 +387,7 @@ range_deserialize(SerializedRanges *serialized)
 	range->nvalues = serialized->nvalues;
 	range->maxvalues = serialized->maxvalues;
 	range->typid = serialized->typid;
+	range->batch_mode = false;
 
 	typbyval = get_typbyval(serialized->typid);
 	typlen = get_typlen(serialized->typid);
@@ -844,11 +857,14 @@ fill_combine_ranges(CombineRange *cranges, int ncranges, Ranges *ranges)
 /*
  * Sort combine ranges using qsort (with BTLessStrategyNumber function).
  */
-static void
+static int
 sort_combine_ranges(FmgrInfo *cmp, Oid colloid,
-					CombineRange *cranges, int ncranges)
+					CombineRange *cranges, int ncranges,
+					bool deduplicate)
 {
-	compare_context cxt;
+	int				n;
+	int				i;
+	compare_context	cxt;
 
 	/* sort the values */
 	cxt.colloid = colloid;
@@ -856,6 +872,24 @@ sort_combine_ranges(FmgrInfo *cmp, Oid colloid,
 
 	qsort_arg(cranges, ncranges, sizeof(CombineRange),
 			  compare_combine_ranges, (void *) &cxt);
+
+	if (!deduplicate)
+		return ncranges;
+
+	/* optionally deduplicate the ranges */
+	n = 1;
+	for (i = 1; i < ncranges; i++)
+	{
+		if (compare_combine_ranges(&cranges[i-1], &cranges[i], (void *) &cxt))
+		{
+			if (i != n)
+				memcpy(&cranges[n], &cranges[i], sizeof(CombineRange));
+
+			n++;
+		}
+	}
+
+	return n;
 }
 
 /*
@@ -1026,26 +1060,40 @@ build_distances(FmgrInfo *distanceFn, Oid colloid,
  */
 static CombineRange *
 build_combine_ranges(FmgrInfo *cmp, Oid colloid, Ranges *ranges,
-					 Datum newvalue, int *nranges)
+					 bool addvalue, Datum newvalue, int *nranges,
+					 bool deduplicate)
 {
 	int				ncranges;
 	CombineRange   *cranges;
 
 	/* now do the actual merge sort */
-	ncranges = ranges->nranges + ranges->nvalues + 1;
+	ncranges = ranges->nranges + ranges->nvalues;
+
+	/* should we add an extra value? */
+	if (addvalue)
+		ncranges += 1;
+
 	cranges = (CombineRange *) palloc0(ncranges * sizeof(CombineRange));
-	*nranges = ncranges;
 
 	/* put the new value at the beginning */
-	cranges[0].minval = newvalue;
-	cranges[0].maxval = newvalue;
-	cranges[0].collapsed = true;
+	if (addvalue)
+	{
+		cranges[0].minval = newvalue;
+		cranges[0].maxval = newvalue;
+		cranges[0].collapsed = true;
 
-	/* then the regular and collapsed ranges */
-	fill_combine_ranges(&cranges[1], ncranges-1, ranges);
+		/* then the regular and collapsed ranges */
+		fill_combine_ranges(&cranges[1], ncranges-1, ranges);
+	}
+	else
+		fill_combine_ranges(cranges, ncranges, ranges);
 
 	/* and sort the ranges */
-	sort_combine_ranges(cmp, colloid, cranges, ncranges);
+	ncranges = sort_combine_ranges(cmp, colloid, cranges, ncranges,
+								   deduplicate);
+
+	/* remember how many cranges we built */
+	*nranges = ncranges;
 
 	return cranges;
 }
@@ -1074,6 +1122,7 @@ count_values(CombineRange *cranges, int ncranges)
 }
 #endif
 
+
 /*
  * reduce_combine_ranges
  *		reduce the ranges until the number of values is low enough
@@ -1250,6 +1299,25 @@ range_add_value(BrinDesc *bdesc, Oid colloid,
 
 	Assert(2*ranges->nranges + ranges->nvalues <= ranges->maxvalues);
 
+	Assert((ranges->nranges >= 0) && (ranges->nvalues >= 0) && (ranges->maxvalues >= 0));
+
+	/*
+	 * When batch-building, there should be no ranges. So either the
+	 * number of ranges is 0 or we're not in batching mode.
+	 */
+	Assert(!ranges->batch_mode || (ranges->nranges == 0));
+
+	/* When batch-building, just add it and we're done. */
+	if (ranges->batch_mode)
+	{
+		/* there has to be free space, if we've sized the struct */
+		Assert(ranges->nvalues < ranges->maxvalues);
+
+		ranges->values[ranges->nvalues++] = newval;
+
+		return true;
+	}
+
 	/*
 	 * Bail out if the value already is covered by the range.
 	 *
@@ -1322,7 +1390,9 @@ range_add_value(BrinDesc *bdesc, Oid colloid,
 					 ranges->nvalues);
 
 	/* OK build the combine ranges */
-	cranges = build_combine_ranges(cmpFn, colloid, ranges, newval, &ncranges);
+	cranges = build_combine_ranges(cmpFn, colloid, ranges,
+								   true, newval, &ncranges,
+								   false);
 
 	/* Reduce the ranges if needed */
 	if (ncranges > ranges->maxvalues)
@@ -1382,6 +1452,81 @@ range_add_value(BrinDesc *bdesc, Oid colloid,
 	return true;
 }
 
+/*
+ * Generate range representation of data collected during "batch mode".
+ * This is similar to reduce_combine_ranges, except that we can't assume
+ * the values are sorted and there may be duplicate values.
+ */
+static void
+compactify_ranges(BrinDesc *bdesc, Ranges *ranges, int max_values)
+{
+	FmgrInfo   *cmpFn,
+			   *distanceFn;
+
+	/* combine ranges */
+	CombineRange   *cranges;
+	int				ncranges;
+	DistanceValue  *distances;
+
+	MemoryContext	ctx;
+	MemoryContext	oldctx;
+
+	/*
+	 * This should only be used in batch mode, and there should be no
+	 * ranges, just individual values.
+	 */
+	Assert((ranges->batch_mode) && (ranges->nranges == 0));
+
+	/* we'll certainly need the comparator, so just look it up now */
+	cmpFn = minmax_multi_get_strategy_procinfo(bdesc, ranges->attno, ranges->typid,
+											   BTLessStrategyNumber);
+
+	/* and we'll also need the 'distance' procedure */
+	distanceFn = minmax_multi_get_procinfo(bdesc, ranges->attno, PROCNUM_DISTANCE);
+
+	/*
+	 * The distanceFn calls (which may internally call e.g. numeric_le) may
+	 * allocate quite a bit of memory, and we must not leak it. Otherwise
+	 * we'd have problems e.g. when building indexes. So we create a local
+	 * memory context and make sure we free the memory before leaving this
+	 * function (not after every call).
+	 */
+	ctx = AllocSetContextCreate(CurrentMemoryContext,
+								"minmax-multi context",
+								ALLOCSET_DEFAULT_SIZES);
+
+	oldctx = MemoryContextSwitchTo(ctx);
+
+	/* OK build the combine ranges */
+	cranges = build_combine_ranges(cmpFn, ranges->colloid, ranges,
+								   false, (Datum) 0, &ncranges,
+								   true);	/* deduplicate */
+
+	/* build array of gap distances and sort them in ascending order */
+	distances = build_distances(distanceFn, ranges->colloid, cranges, ncranges);
+
+	/*
+	 * Combine ranges until we get below max_values. We don't use any scale
+	 * factor, because this is used at the very end of "batch mode" and we
+	 * don't expect more tuples to be inserted soon.
+	 */
+	ncranges = reduce_combine_ranges(cranges, ncranges, distances,
+									  max_values, cmpFn, ranges->colloid);
+
+	Assert(count_values(cranges, ncranges) <= max_values);
+
+	/* decompose the combine ranges into regular ranges and single values */
+	store_combine_ranges(ranges, cranges, ncranges);
+
+	MemoryContextSwitchTo(oldctx);
+	MemoryContextDelete(ctx);
+
+	/* Check the ordering invariants are not violated (for both parts). */
+	AssertArrayOrder(cmpFn, ranges->colloid, ranges->values, ranges->nranges*2);
+	AssertArrayOrder(cmpFn, ranges->colloid, &ranges->values[ranges->nranges*2],
+					 ranges->nvalues);
+}
+
 Datum
 brin_minmax_multi_opcinfo(PG_FUNCTION_ARGS)
 {
@@ -1799,10 +1944,16 @@ brin_minmax_multi_distance_inet(PG_FUNCTION_ARGS)
 }
 
 static void
-brin_minmax_multi_serialize(Datum src, Datum *dst)
+brin_minmax_multi_serialize(BrinDesc *bdesc, Datum src, Datum *dst)
 {
 	Ranges *ranges = (Ranges *) DatumGetPointer(src);
-	SerializedRanges *s = range_serialize(ranges);
+	SerializedRanges *s;
+
+	/* In batch mode, we need to compress the accumulated values. */
+	if (ranges->batch_mode)
+		compactify_ranges(bdesc, ranges, ranges->target_maxvalues);
+
+	s = range_serialize(ranges);
 	dst[0] = PointerGetDatum(s);
 }
 
@@ -1845,15 +1996,28 @@ brin_minmax_multi_add_value(PG_FUNCTION_ARGS)
 	/*
 	 * If this is the first non-null value, we need to initialize the range
 	 * list. Otherwise just extract the existing range list from BrinValues.
+	 *
+	 * When starting with an empty range, we assume this is a batch mode,
+	 * i.e. we size the buffer for the maximum possible number of items in
+	 * the range (based on range size and max number of items on a page).
+	 *
+	 * XXX This may require quite a bit of memory, so maybe we should use
+	 * some value in between. OTOH most tables will have much wider rows,
+	 * so the number of rows per page is much lower.
 	 */
 	if (column->bv_allnulls)
 	{
 		MemoryContext oldctx;
 
-		oldctx = MemoryContextSwitchTo(column->bv_context);
+		BlockNumber		pagesPerRange = BrinGetPagesPerRange(bdesc->bd_index);
 
-		ranges = minmax_multi_init(brin_minmax_multi_get_values(bdesc, opts));
+		oldctx = MemoryContextSwitchTo(column->bv_context);
+		ranges = minmax_multi_init(MaxHeapTuplesPerPage * pagesPerRange);
+		ranges->attno = attno;
+		ranges->colloid = colloid;
 		ranges->typid = attr->atttypid;
+		ranges->batch_mode = true;
+		ranges->target_maxvalues = brin_minmax_multi_get_values(bdesc, opts);
 
 		MemoryContextSwitchTo(oldctx);
 
@@ -2125,7 +2289,7 @@ brin_minmax_multi_union(PG_FUNCTION_ARGS)
 	oldctx = MemoryContextSwitchTo(ctx);
 
 	/* allocate and fill */
-	cranges = (CombineRange *)palloc0(ncranges * sizeof(CombineRange));
+	cranges = (CombineRange *) palloc0(ncranges * sizeof(CombineRange));
 
 	/* fill the combine ranges with entries for the first range */
 	fill_combine_ranges(cranges, ranges_a->nranges + ranges_a->nvalues,
@@ -2139,8 +2303,8 @@ brin_minmax_multi_union(PG_FUNCTION_ARGS)
 	cmpFn = minmax_multi_get_strategy_procinfo(bdesc, attno, attr->atttypid,
 											 BTLessStrategyNumber);
 
-	/* sort the combine ranges */
-	sort_combine_ranges(cmpFn, colloid, cranges, ncranges);
+	/* sort the combine ranges (don't deduplicate) */
+	sort_combine_ranges(cmpFn, colloid, cranges, ncranges, false);
 
 	/*
 	 * We've merged two different lists of ranges, so some of them may be
diff --git a/src/backend/access/brin/brin_tuple.c b/src/backend/access/brin/brin_tuple.c
index df820db029..bf8635d788 100644
--- a/src/backend/access/brin/brin_tuple.c
+++ b/src/backend/access/brin/brin_tuple.c
@@ -163,6 +163,7 @@ brin_form_tuple(BrinDesc *brdesc, BlockNumber blkno, BrinMemTuple *tuple,
 		if (tuple->bt_columns[keyno].bv_serialize)
 		{
 			tuple->bt_columns[keyno].bv_serialize(
+				brdesc,
 				tuple->bt_columns[keyno].bv_mem_value,
 				tuple->bt_columns[keyno].bv_values);
 		}
diff --git a/src/include/access/brin_tuple.h b/src/include/access/brin_tuple.h
index 66a6c3c792..e08998b71a 100644
--- a/src/include/access/brin_tuple.h
+++ b/src/include/access/brin_tuple.h
@@ -14,7 +14,7 @@
 #include "access/brin_internal.h"
 #include "access/tupdesc.h"
 
-typedef void (*brin_serialize_callback_type) (Datum src, Datum * dst);
+typedef void (*brin_serialize_callback_type) (BrinDesc *bdesc, Datum src, Datum * dst);
 
 /*
  * A BRIN index stores one index tuple per page range.  Each index tuple
-- 
2.26.2


--------------FA974B75A0C3E9CA18CE7207
Content-Type: application/x-shellscript;
 name="brin.sh"
Content-Transfer-Encoding: base64
Content-Disposition: attachment;
 filename="brin.sh"

IyEvdXNyL2Jpbi9iYXNoCgpOUk9XUz0kMQoKZHJvcGRiIC0taWYtZXhpc3RzIHRlc3QKY3Jl
YXRlZGIgdGVzdAoKIyBtb25vdG9uaWMsIGFzY2VuZGluZywgaW50NAoKcHNxbCB0ZXN0IC1j
ICJjcmVhdGUgdW5sb2dnZWQgdGFibGUgdCAoYSBpbnQpIiA+IC9kZXYvbnVsbCAyPiYxCnBz
cWwgdGVzdCAtYyAiaW5zZXJ0IGludG8gdCBzZWxlY3QgaSBmcm9tIGdlbmVyYXRlX3Nlcmll
cygxLCROUk9XUykgcyhpKSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgInZhY3V1
bSBhbmFseXplIHQiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJjaGVja3BvaW50
IiA+IC9kZXYvbnVsbCAyPiYxCgpmb3IgciBpbiBgc2VxIDEgMTBgOyBkbwoKCSMgcmVndWxh
ciBidHJlZSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVh
dGUgaW5kZXggaWR4IG9uIHQgKGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBU
aW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1v
bm90b25pYy1hc2MgaW50NCBidHJlZSAkciAkdAoKCSMgYnJpbiBtaW5tYXggaW5kZXgKCXBz
cWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0
IHVzaW5nIGJyaW4gKGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJy
aW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25p
Yy1hc2MgaW50NCBicmluLW1pbm1heCAkciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5k
ZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlk
eCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRl
eCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAn
e3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtYXNjIGludDQgYnJpbi1tdWx0aS1taW5t
YXggJHIgJHQKCgkjIGJyaW4gYmxvb20gaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8
RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50NF9i
bG9vbV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1w
IHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy1hc2Mg
aW50NCBicmluLWJsb29tICRyICR0Cgpkb25lCgpwc3FsIHRlc3QgLWMgImRyb3AgdGFibGUg
dCIgPiAvZGV2L251bGwgMj4mMQoKCiMgbW9ub3RvbmljLCBhc2NlbmRpbmcsIGludDQsIHBh
ZGRlZAoKcHNxbCB0ZXN0IC1jICJjcmVhdGUgdW5sb2dnZWQgdGFibGUgdCAoYSBpbnQsIGIg
dGV4dCkiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJpbnNlcnQgaW50byB0IHNl
bGVjdCBpLCBtZDUoaTo6dGV4dCkgZnJvbSBnZW5lcmF0ZV9zZXJpZXMoMSwkTlJPV1MpIHMo
aSkiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJ2YWN1dW0gYW5hbHl6ZSB0IiA+
IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAiY2hlY2twb2ludCIgPiAvZGV2L251bGwg
Mj4mMQoKZm9yIHIgaW4gYHNlcSAxIDEwYDsgZG8KCgkjIHJlZ3VsYXIgYnRyZWUgaW5kZXgK
CXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBv
biB0IChhKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8
IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtYXNjIGlu
dDQtcGFkZGVkIGJ0cmVlICRyICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJcHNxbCB0ZXN0
ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcg
YnJpbiAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAg
fCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLWFzYyBp
bnQ0LXBhZGRlZCBicmluLW1pbm1heCAkciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5k
ZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlk
eCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRl
eCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAn
e3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtYXNjIGludDQtcGFkZGVkIGJyaW4tbXVs
dGktbWlubWF4ICRyICR0CgoJIyBicmluIGJsb29tIGluZGV4Cglwc3FsIHRlc3QgPiBicmlu
LnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1c2luZyBicmluIChh
IGludDRfYmxvb21fb3BzKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBi
cmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9u
aWMtYXNjIGludDQtcGFkZGVkIGJyaW4tYmxvb20gJHIgJHQKCmRvbmUKCnBzcWwgdGVzdCAt
YyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVsbCAyPiYxCgoKIyBtb25vdG9uaWMsIGRlc2Nl
bmRpbmcsIGludDQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9nZ2VkIHRhYmxlIHQgKGEg
aW50KSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImluc2VydCBpbnRvIHQgc2Vs
ZWN0IGkgZnJvbSBnZW5lcmF0ZV9zZXJpZXMoJE5ST1dTLDEsLTEpIHMoaSkiID4gL2Rldi9u
dWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJ2YWN1dW0gYW5hbHl6ZSB0IiA+IC9kZXYvbnVsbCAy
PiYxCnBzcWwgdGVzdCAtYyAiY2hlY2twb2ludCIgPiAvZGV2L251bGwgMj4mMQoKZm9yIHIg
aW4gYHNlcSAxIDEwYDsgZG8KCgkjIHJlZ3VsYXIgYnRyZWUgaW5kZXgKCXBzcWwgdGVzdCA+
IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IChhKTsKZHJv
cCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8
IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtZGVzYyBpbnQ0IGJ0cmVlICRy
ICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YK
XHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSk7CmRyb3AgaW5k
ZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sg
J3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLWRlc2MgaW50NCBicmluLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1w
IDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50
NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGlt
ZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25v
dG9uaWMtZGVzYyBpbnQ0IGJyaW4tbXVsdGktbWlubWF4ICRyICR0CgoJIyBicmluIGJsb29t
IGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRl
eCBpZHggb24gdCB1c2luZyBicmluIChhIGludDRfYmxvb21fb3BzKTsKZHJvcCBpbmRleCBp
ZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3By
aW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtZGVzYyBpbnQ0IGJyaW4tYmxvb20gJHIgJHQK
CmRvbmUKCnBzcWwgdGVzdCAtYyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVsbCAyPiYxCgoK
IyBtb25vdG9uaWMsIGRlc2NlbmRpbmcsIGludDQsIHBhZGRlZAoKcHNxbCB0ZXN0IC1jICJj
cmVhdGUgdW5sb2dnZWQgdGFibGUgdCAoYSBpbnQsIGIgdGV4dCkiID4gL2Rldi9udWxsIDI+
JjEKcHNxbCB0ZXN0IC1jICJpbnNlcnQgaW50byB0IHNlbGVjdCBpLCBtZDUoaTo6dGV4dCkg
ZnJvbSBnZW5lcmF0ZV9zZXJpZXMoJE5ST1dTLDEsLTEpIHMoaSkiID4gL2Rldi9udWxsIDI+
JjEKcHNxbCB0ZXN0IC1jICJ2YWN1dW0gYW5hbHl6ZSB0IiA+IC9kZXYvbnVsbCAyPiYxCnBz
cWwgdGVzdCAtYyAiY2hlY2twb2ludCIgPiAvZGV2L251bGwgMj4mMQoKZm9yIHIgaW4gYHNl
cSAxIDEwYDsgZG8KCgkjIHJlZ3VsYXIgYnRyZWUgaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4u
dG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IChhKTsKZHJvcCBpbmRl
eCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAn
e3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtZGVzYyBpbnQ0LXBhZGRlZCBidHJlZSAk
ciAkdAoKCSMgYnJpbiBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9G
Clx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEpOwpkcm9wIGlu
ZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdr
ICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy1kZXNjIGludDQtcGFkZGVkIGJyaW4t
bWlubWF4ICRyICR0CgoJIyBicmluIG11bHRpIG1pbm1heCBpbmRleAoJcHNxbCB0ZXN0ID4g
YnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJp
biAoYSBpbnQ0X21pbm1heF9tdWx0aV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1g
Z3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgll
Y2hvIG1vbm90b25pYy1kZXNjIGludDQtcGFkZGVkIGJyaW4tbXVsdGktbWlubWF4ICRyICR0
CgoJIyBicmluIGJsb29tIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGlt
aW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1c2luZyBicmluIChhIGludDRfYmxvb21fb3Bz
KTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQg
LW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtZGVzYyBpbnQ0LXBh
ZGRlZCBicmluLWJsb29tICRyICR0Cgpkb25lCgpwc3FsIHRlc3QgLWMgImRyb3AgdGFibGUg
dCIgPiAvZGV2L251bGwgMj4mMQoKIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMj
IyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjCgojIG1vbm90b25pYywg
MTAwIGRpc3RpbmN0IHZhbHVlcywgaW50NAoKcHNxbCB0ZXN0IC1jICJjcmVhdGUgdW5sb2dn
ZWQgdGFibGUgdCAoYSBpbnQpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAiaW5z
ZXJ0IGludG8gdCBzZWxlY3QgbW9kKGksMTAwKSBmcm9tIGdlbmVyYXRlX3NlcmllcygxLCRO
Uk9XUykgcyhpKSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgInZhY3V1bSBhbmFs
eXplIHQiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJjaGVja3BvaW50IiA+IC9k
ZXYvbnVsbCAyPiYxCgpmb3IgciBpbiBgc2VxIDEgMTBgOyBkbwoKCSMgcmVndWxhciBidHJl
ZSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5k
ZXggaWR4IG9uIHQgKGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJy
aW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25p
Yy0xMDAtYXNjIGludDQgYnRyZWUgJHIgJHQKCgkjIGJyaW4gbWlubWF4IGluZGV4Cglwc3Fs
IHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1
c2luZyBicmluIChhKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmlu
LnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMt
MTAwLWFzYyBpbnQ0IGJyaW4tbWlubWF4ICRyICR0CgoJIyBicmluIG11bHRpIG1pbm1heCBp
bmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXgg
aWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X21pbm1heF9tdWx0aV9vcHMpOwpkcm9wIGlu
ZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdr
ICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAtYXNjIGludDQgYnJpbi1tdWx0
aS1taW5tYXggJHIgJHQKCgkjIGJyaW4gYmxvb20gaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4u
dG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEg
aW50NF9ibG9vbV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJy
aW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25p
Yy0xMDAtYXNjIGludDQgYnJpbi1ibG9vbSAkciAkdAoKZG9uZQoKcHNxbCB0ZXN0IC1jICJk
cm9wIHRhYmxlIHQiID4gL2Rldi9udWxsIDI+JjEKCgojIG1vbm90b25pYywgMTAwIHZhbHVl
cywgYXNjZW5kaW5nLCBpbnQ0LCBwYWRkZWQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9n
Z2VkIHRhYmxlIHQgKGEgaW50LCBiIHRleHQpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVz
dCAtYyAiaW5zZXJ0IGludG8gdCBzZWxlY3QgbW9kKGksMTAwKSwgbWQ1KGk6OnRleHQpIGZy
b20gZ2VuZXJhdGVfc2VyaWVzKDEsJE5ST1dTKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBz
cWwgdGVzdCAtYyAidmFjdXVtIGFuYWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRl
c3QgLWMgImNoZWNrcG9pbnQiID4gL2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAx
MGA7IGRvCgoJIyByZWd1bGFyIGJ0cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8
PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4
OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmlu
dCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMC1hc2MgaW50NC1wYWRkZWQgYnRyZWUgJHIg
JHQKCgkjIGJyaW4gbWlubWF4IGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpc
dGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1c2luZyBicmluIChhKTsKZHJvcCBpbmRl
eCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAn
e3ByaW50ICQyfSdgCgoJZWNobyBtb25vdG9uaWMtMTAwLWFzYyBpbnQ0LXBhZGRlZCBicmlu
LW1pbm1heCAkciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+
IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJy
aW4gKGEgaW50NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9
YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJ
ZWNobyBtb25vdG9uaWMtMTAwLWFzYyBpbnQ0LXBhZGRlZCBicmluLW11bHRpLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YK
XHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29t
X29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMC1hc2Mg
aW50NC1wYWRkZWQgYnJpbi1ibG9vbSAkciAkdAoKZG9uZQoKcHNxbCB0ZXN0IC1jICJkcm9w
IHRhYmxlIHQiID4gL2Rldi9udWxsIDI+JjEKCgojIG1vbm90b25pYywgMTAwIGRpc3RpbmN0
IHZhbHVlcywgZGVzY2VuZGluZywgaW50NAoKcHNxbCB0ZXN0IC1jICJjcmVhdGUgdW5sb2dn
ZWQgdGFibGUgdCAoYSBpbnQpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAiaW5z
ZXJ0IGludG8gdCBzZWxlY3QgbW9kKGksMTAwKSBmcm9tIGdlbmVyYXRlX3NlcmllcygkTlJP
V1MsMSwtMSkgcyhpKSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgInZhY3V1bSBh
bmFseXplIHQiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJjaGVja3BvaW50IiA+
IC9kZXYvbnVsbCAyPiYxCgpmb3IgciBpbiBgc2VxIDEgMTBgOyBkbwoKCSMgcmVndWxhciBi
dHJlZSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUg
aW5kZXggaWR4IG9uIHQgKGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1l
IGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90
b25pYy0xMDAtZGVzYyBpbnQ0IGJ0cmVlICRyICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJ
cHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9u
IHQgdXNpbmcgYnJpbiAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUg
YnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3Rv
bmljLTEwMC1kZXNjIGludDQgYnJpbi1taW5tYXggJHIgJHQKCgkjIGJyaW4gbXVsdGkgbWlu
bWF4IGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBp
bmRleCBpZHggb24gdCB1c2luZyBicmluIChhIGludDRfbWlubWF4X211bHRpX29wcyk7CmRy
b3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEg
fCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMC1kZXNjIGludDQgYnJp
bi1tdWx0aS1taW5tYXggJHIgJHQKCgkjIGJyaW4gYmxvb20gaW5kZXgKCXBzcWwgdGVzdCA+
IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJy
aW4gKGEgaW50NF9ibG9vbV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBU
aW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1v
bm90b25pYy0xMDAtZGVzYyBpbnQ0IGJyaW4tYmxvb20gJHIgJHQKCmRvbmUKCnBzcWwgdGVz
dCAtYyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVsbCAyPiYxCgoKIyBtb25vdG9uaWMsIDEw
MCBkaXN0aW5jdCB2YWx1ZXMsIGRlc2NlbmRpbmcsIGludDQsIHBhZGRlZAoKcHNxbCB0ZXN0
IC1jICJjcmVhdGUgdW5sb2dnZWQgdGFibGUgdCAoYSBpbnQsIGIgdGV4dCkiID4gL2Rldi9u
dWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJpbnNlcnQgaW50byB0IHNlbGVjdCBtb2QoaSwgMTAw
KSwgbWQ1KGk6OnRleHQpIGZyb20gZ2VuZXJhdGVfc2VyaWVzKCROUk9XUywxLC0xKSBzKGkp
IiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFjdXVtIGFuYWx5emUgdCIgPiAv
ZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9pbnQiID4gL2Rldi9udWxsIDI+
JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1bGFyIGJ0cmVlIGluZGV4Cglw
c3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24g
dCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMC1kZXNj
IGludDQtcGFkZGVkIGJ0cmVlICRyICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJcHNxbCB0
ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNp
bmcgYnJpbiAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50
bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEw
MC1kZXNjIGludDQtcGFkZGVkIGJyaW4tbWlubWF4ICRyICR0CgoJIyBicmluIG11bHRpIG1p
bm1heCBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUg
aW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X21pbm1heF9tdWx0aV9vcHMpOwpk
cm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAx
IHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAtZGVzYyBpbnQ0LXBh
ZGRlZCBicmluLW11bHRpLW1pbm1heCAkciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNx
bCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQg
dXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29tX29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0
PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoK
CWVjaG8gbW9ub3RvbmljLTEwMC1kZXNjIGludDQtcGFkZGVkIGJyaW4tYmxvb20gJHIgJHQK
CmRvbmUKCnBzcWwgdGVzdCAtYyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVsbCAyPiYxCgoj
IyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMj
IyMjIyMjIyMjIyMjIyMjIyMKCiMgbW9ub3RvbmljLCAxMDAwMCBkaXN0aW5jdCB2YWx1ZXMs
IGludDQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9nZ2VkIHRhYmxlIHQgKGEgaW50KSIg
PiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImluc2VydCBpbnRvIHQgc2VsZWN0IG1v
ZChpLDEwMDAwKSBmcm9tIGdlbmVyYXRlX3NlcmllcygxLCROUk9XUykgcyhpKSIgPiAvZGV2
L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgInZhY3V1bSBhbmFseXplIHQiID4gL2Rldi9udWxs
IDI+JjEKcHNxbCB0ZXN0IC1jICJjaGVja3BvaW50IiA+IC9kZXYvbnVsbCAyPiYxCgpmb3Ig
ciBpbiBgc2VxIDEgMTBgOyBkbwoKCSMgcmVndWxhciBidHJlZSBpbmRleAoJcHNxbCB0ZXN0
ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgKGEpOwpk
cm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAx
IHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1hc2MgaW50NCBi
dHJlZSAkciAkdAoKCSMgYnJpbiBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1w
IDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEpOwpk
cm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAx
IHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1hc2MgaW50NCBi
cmluLW1pbm1heCAkciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5kZXgKCXBzcWwgdGVz
dCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5n
IGJyaW4gKGEgaW50NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoK
CXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdg
CgoJZWNobyBtb25vdG9uaWMtMTAwMDAtYXNjIGludDQgYnJpbi1tdWx0aS1taW5tYXggJHIg
JHQKCgkjIGJyaW4gYmxvb20gaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0
aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50NF9ibG9vbV9v
cHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVh
ZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1hc2Mg
aW50NCBicmluLWJsb29tICRyICR0Cgpkb25lCgpwc3FsIHRlc3QgLWMgImRyb3AgdGFibGUg
dCIgPiAvZGV2L251bGwgMj4mMQoKCiMgbW9ub3RvbmljLCAxMDAwMCB2YWx1ZXMsIGFzY2Vu
ZGluZywgaW50NCwgcGFkZGVkCgpwc3FsIHRlc3QgLWMgImNyZWF0ZSB1bmxvZ2dlZCB0YWJs
ZSB0IChhIGludCwgYiB0ZXh0KSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImlu
c2VydCBpbnRvIHQgc2VsZWN0IG1vZChpLDEwMDAwKSwgbWQ1KGk6OnRleHQpIGZyb20gZ2Vu
ZXJhdGVfc2VyaWVzKDEsJE5ST1dTKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVz
dCAtYyAidmFjdXVtIGFuYWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMg
ImNoZWNrcG9pbnQiID4gL2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRv
CgoJIyByZWd1bGFyIGJ0cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpc
dGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YK
Cgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0n
YAoKCWVjaG8gbW9ub3RvbmljLTEwMDAwLWFzYyBpbnQ0LXBhZGRlZCBidHJlZSAkciAkdAoK
CSMgYnJpbiBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1p
bmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEpOwpkcm9wIGluZGV4IGlk
eDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJp
bnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1hc2MgaW50NC1wYWRkZWQgYnJpbi1t
aW5tYXggJHIgJHQKCgkjIGJyaW4gbXVsdGkgbWlubWF4IGluZGV4Cglwc3FsIHRlc3QgPiBi
cmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1c2luZyBicmlu
IChhIGludDRfbWlubWF4X211bHRpX29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBn
cmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVj
aG8gbW9ub3RvbmljLTEwMDAwLWFzYyBpbnQ0LXBhZGRlZCBicmluLW11bHRpLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YK
XHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29t
X29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMDAwLWFz
YyBpbnQ0LXBhZGRlZCBicmluLWJsb29tICRyICR0Cgpkb25lCgpwc3FsIHRlc3QgLWMgImRy
b3AgdGFibGUgdCIgPiAvZGV2L251bGwgMj4mMQoKCiMgbW9ub3RvbmljLCAxMDAwMCBkaXN0
aW5jdCB2YWx1ZXMsIGRlc2NlbmRpbmcsIGludDQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVu
bG9nZ2VkIHRhYmxlIHQgKGEgaW50KSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMg
Imluc2VydCBpbnRvIHQgc2VsZWN0IG1vZChpLDEwMDAwKSBmcm9tIGdlbmVyYXRlX3Nlcmll
cygkTlJPV1MsMSwtMSkgcyhpKSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgInZh
Y3V1bSBhbmFseXplIHQiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJjaGVja3Bv
aW50IiA+IC9kZXYvbnVsbCAyPiYxCgpmb3IgciBpbiBgc2VxIDEgMTBgOyBkbwoKCSMgcmVn
dWxhciBidHJlZSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpj
cmVhdGUgaW5kZXggaWR4IG9uIHQgKGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3Jl
cCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hv
IG1vbm90b25pYy0xMDAwMC1kZXNjIGludDQgYnRyZWUgJHIgJHQKCgkjIGJyaW4gbWlubWF4
IGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRl
eCBpZHggb24gdCB1c2luZyBicmluIChhKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdy
ZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNo
byBtb25vdG9uaWMtMTAwMDAtZGVzYyBpbnQ0IGJyaW4tbWlubWF4ICRyICR0CgoJIyBicmlu
IG11bHRpIG1pbm1heCBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWlu
ZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X21pbm1heF9tdWx0
aV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwg
aGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1k
ZXNjIGludDQgYnJpbi1tdWx0aS1taW5tYXggJHIgJHQKCgkjIGJyaW4gYmxvb20gaW5kZXgK
CXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBv
biB0IHVzaW5nIGJyaW4gKGEgaW50NF9ibG9vbV9vcHMpOwpkcm9wIGluZGV4IGlkeDsKRU9G
CgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9
J2AKCgllY2hvIG1vbm90b25pYy0xMDAwMC1kZXNjIGludDQgYnJpbi1ibG9vbSAkciAkdAoK
ZG9uZQoKcHNxbCB0ZXN0IC1jICJkcm9wIHRhYmxlIHQiID4gL2Rldi9udWxsIDI+JjEKCgoj
IG1vbm90b25pYywgMTAwMDAgZGlzdGluY3QgdmFsdWVzLCBkZXNjZW5kaW5nLCBpbnQ0LCBw
YWRkZWQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9nZ2VkIHRhYmxlIHQgKGEgaW50LCBi
IHRleHQpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAiaW5zZXJ0IGludG8gdCBz
ZWxlY3QgbW9kKGksIDEwMDAwKSwgbWQ1KGk6OnRleHQpIGZyb20gZ2VuZXJhdGVfc2VyaWVz
KCROUk9XUywxLC0xKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFj
dXVtIGFuYWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9p
bnQiID4gL2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1
bGFyIGJ0cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNy
ZWF0ZSBpbmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVw
IFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8g
bW9ub3RvbmljLTEwMDAwLWRlc2MgaW50NC1wYWRkZWQgYnRyZWUgJHIgJHQKCgkjIGJyaW4g
bWlubWF4IGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0
ZSBpbmRleCBpZHggb24gdCB1c2luZyBicmluIChhKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoK
CXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdg
CgoJZWNobyBtb25vdG9uaWMtMTAwMDAtZGVzYyBpbnQ0LXBhZGRlZCBicmluLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJyaW4udG1w
IDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4gKGEgaW50
NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGlt
ZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyBtb25v
dG9uaWMtMTAwMDAtZGVzYyBpbnQ0LXBhZGRlZCBicmluLW11bHRpLW1pbm1heCAkciAkdAoK
CSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWlu
ZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29tX29wcyk7
CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1u
IDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gbW9ub3RvbmljLTEwMDAwLWRlc2MgaW50
NC1wYWRkZWQgYnJpbi1ibG9vbSAkciAkdAoKZG9uZQoKcHNxbCB0ZXN0IC1jICJkcm9wIHRh
YmxlIHQiID4gL2Rldi9udWxsIDI+JjEKCiMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMj
IyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIwoKIyByYW5kb20s
IDEwMCBkaXN0aW5jdCB2YWx1ZXMsIGludDQKCnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9n
Z2VkIHRhYmxlIHQgKGEgaW50KSIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImlu
c2VydCBpbnRvIHQgc2VsZWN0IDEwMCAqIHJhbmRvbSgpIGZyb20gZ2VuZXJhdGVfc2VyaWVz
KDEsJE5ST1dTKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFjdXVt
IGFuYWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9pbnQi
ID4gL2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1bGFy
IGJ0cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0
ZSBpbmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRp
bWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFu
ZG9tLTEwMCBpbnQ0IGJ0cmVlICRyICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJcHNxbCB0
ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNp
bmcgYnJpbiAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50
bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9tLTEwMCBp
bnQ0IGJyaW4tbWlubWF4ICRyICR0CgoJIyBicmluIG11bHRpIG1pbm1heCBpbmRleAoJcHNx
bCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQg
dXNpbmcgYnJpbiAoYSBpbnQ0X21pbm1heF9tdWx0aV9vcHMpOwpkcm9wIGluZGV4IGlkeDsK
RU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAxIHwgYXdrICd7cHJpbnQg
JDJ9J2AKCgllY2hvIHJhbmRvbS0xMDAgaW50NCBicmluLW11bHRpLW1pbm1heCAkciAkdAoK
CSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWlu
ZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29tX29wcyk7
CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1u
IDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9tLTEwMCBpbnQ0IGJyaW4tYmxv
b20gJHIgJHQKCmRvbmUKCnBzcWwgdGVzdCAtYyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVs
bCAyPiYxCgoKIyByYW5kb20sIDEwMCB2YWx1ZXMsIGludDQsIHBhZGRlZAoKcHNxbCB0ZXN0
IC1jICJjcmVhdGUgdW5sb2dnZWQgdGFibGUgdCAoYSBpbnQsIGIgdGV4dCkiID4gL2Rldi9u
dWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJpbnNlcnQgaW50byB0IHNlbGVjdCAxMDAgKiByYW5k
b20oKSwgbWQ1KGk6OnRleHQpIGZyb20gZ2VuZXJhdGVfc2VyaWVzKDEsJE5ST1dTKSBzKGkp
IiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFjdXVtIGFuYWx5emUgdCIgPiAv
ZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9pbnQiID4gL2Rldi9udWxsIDI+
JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1bGFyIGJ0cmVlIGluZGV4Cglw
c3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24g
dCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9tLTEwMCBpbnQ0LXBh
ZGRlZCBidHJlZSAkciAkdAoKCSMgYnJpbiBtaW5tYXggaW5kZXgKCXBzcWwgdGVzdCA+IGJy
aW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBvbiB0IHVzaW5nIGJyaW4g
KGEpOwpkcm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVh
ZCAtbiAxIHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIHJhbmRvbS0xMDAgaW50NC1wYWRk
ZWQgYnJpbi1taW5tYXggJHIgJHQKCgkjIGJyaW4gbXVsdGkgbWlubWF4IGluZGV4Cglwc3Fs
IHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1
c2luZyBicmluIChhIGludDRfbWlubWF4X211bHRpX29wcyk7CmRyb3AgaW5kZXggaWR4OwpF
T0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAk
Mn0nYAoKCWVjaG8gcmFuZG9tLTEwMCBpbnQ0LXBhZGRlZCBicmluLW11bHRpLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YK
XHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29t
X29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9tLTEwMCBpbnQ0LXBh
ZGRlZCBicmluLWJsb29tICRyICR0Cgpkb25lCgpwc3FsIHRlc3QgLWMgImRyb3AgdGFibGUg
dCIgPiAvZGV2L251bGwgMj4mMQoKIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMj
IyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjIyMjCgojIHJhbmRvbSwgMTAw
MDAgZGlzdGluY3QgdmFsdWVzLCBpbnQ0Cgpwc3FsIHRlc3QgLWMgImNyZWF0ZSB1bmxvZ2dl
ZCB0YWJsZSB0IChhIGludCkiID4gL2Rldi9udWxsIDI+JjEKcHNxbCB0ZXN0IC1jICJpbnNl
cnQgaW50byB0IHNlbGVjdCAxMDAwMCAqIHJhbmRvbSgpIGZyb20gZ2VuZXJhdGVfc2VyaWVz
KDEsJE5ST1dTKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFjdXVt
IGFuYWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9pbnQi
ID4gL2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1bGFy
IGJ0cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0
ZSBpbmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRp
bWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFu
ZG9tLTEwMDAwIGludDQgYnRyZWUgJHIgJHQKCgkjIGJyaW4gbWlubWF4IGluZGV4Cglwc3Fs
IHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBpbmRleCBpZHggb24gdCB1
c2luZyBicmluIChhKTsKZHJvcCBpbmRleCBpZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmlu
LnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3ByaW50ICQyfSdgCgoJZWNobyByYW5kb20tMTAw
MDAgaW50NCBicmluLW1pbm1heCAkciAkdAoKCSMgYnJpbiBtdWx0aSBtaW5tYXggaW5kZXgK
CXBzcWwgdGVzdCA+IGJyaW4udG1wIDw8RU9GClx0aW1pbmcKY3JlYXRlIGluZGV4IGlkeCBv
biB0IHVzaW5nIGJyaW4gKGEgaW50NF9taW5tYXhfbXVsdGlfb3BzKTsKZHJvcCBpbmRleCBp
ZHg7CkVPRgoKCXQ9YGdyZXAgVGltZSBicmluLnRtcCB8IGhlYWQgLW4gMSB8IGF3ayAne3By
aW50ICQyfSdgCgoJZWNobyByYW5kb20tMTAwMDAgaW50NCBicmluLW11bHRpLW1pbm1heCAk
ciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YK
XHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X2Jsb29t
X29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUgYnJpbi50bXAgfCBo
ZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9tLTEwMDAwIGludDQg
YnJpbi1ibG9vbSAkciAkdAoKZG9uZQoKcHNxbCB0ZXN0IC1jICJkcm9wIHRhYmxlIHQiID4g
L2Rldi9udWxsIDI+JjEKCgojIHJhbmRvbSwgMTAwMDAgdmFsdWVzLCBpbnQ0LCBwYWRkZWQK
CnBzcWwgdGVzdCAtYyAiY3JlYXRlIHVubG9nZ2VkIHRhYmxlIHQgKGEgaW50LCBiIHRleHQp
IiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAiaW5zZXJ0IGludG8gdCBzZWxlY3Qg
MTAwMDAgKiByYW5kb20oKSwgbWQ1KGk6OnRleHQpIGZyb20gZ2VuZXJhdGVfc2VyaWVzKDEs
JE5ST1dTKSBzKGkpIiA+IC9kZXYvbnVsbCAyPiYxCnBzcWwgdGVzdCAtYyAidmFjdXVtIGFu
YWx5emUgdCIgPiAvZGV2L251bGwgMj4mMQpwc3FsIHRlc3QgLWMgImNoZWNrcG9pbnQiID4g
L2Rldi9udWxsIDI+JjEKCmZvciByIGluIGBzZXEgMSAxMGA7IGRvCgoJIyByZWd1bGFyIGJ0
cmVlIGluZGV4Cglwc3FsIHRlc3QgPiBicmluLnRtcCA8PEVPRgpcdGltaW5nCmNyZWF0ZSBp
bmRleCBpZHggb24gdCAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUg
YnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9t
LTEwMDAwIGludDQtcGFkZGVkIGJ0cmVlICRyICR0CgoJIyBicmluIG1pbm1heCBpbmRleAoJ
cHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9u
IHQgdXNpbmcgYnJpbiAoYSk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVwIFRpbWUg
YnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8gcmFuZG9t
LTEwMDAwIGludDQtcGFkZGVkIGJyaW4tbWlubWF4ICRyICR0CgoJIyBicmluIG11bHRpIG1p
bm1heCBpbmRleAoJcHNxbCB0ZXN0ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUg
aW5kZXggaWR4IG9uIHQgdXNpbmcgYnJpbiAoYSBpbnQ0X21pbm1heF9tdWx0aV9vcHMpOwpk
cm9wIGluZGV4IGlkeDsKRU9GCgoJdD1gZ3JlcCBUaW1lIGJyaW4udG1wIHwgaGVhZCAtbiAx
IHwgYXdrICd7cHJpbnQgJDJ9J2AKCgllY2hvIHJhbmRvbS0xMDAwMCBpbnQ0LXBhZGRlZCBi
cmluLW11bHRpLW1pbm1heCAkciAkdAoKCSMgYnJpbiBibG9vbSBpbmRleAoJcHNxbCB0ZXN0
ID4gYnJpbi50bXAgPDxFT0YKXHRpbWluZwpjcmVhdGUgaW5kZXggaWR4IG9uIHQgdXNpbmcg
YnJpbiAoYSBpbnQ0X2Jsb29tX29wcyk7CmRyb3AgaW5kZXggaWR4OwpFT0YKCgl0PWBncmVw
IFRpbWUgYnJpbi50bXAgfCBoZWFkIC1uIDEgfCBhd2sgJ3twcmludCAkMn0nYAoKCWVjaG8g
cmFuZG9tLTEwMDAwIGludDQtcGFkZGVkIGJyaW4tYmxvb20gJHIgJHQKCmRvbmUKCnBzcWwg
dGVzdCAtYyAiZHJvcCB0YWJsZSB0IiA+IC9kZXYvbnVsbCAyPiYxCgo=
--------------FA974B75A0C3E9CA18CE7207--





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-10-21 21:24  Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-10-21 21:24 UTC (permalink / raw)
  To: Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

Hi,

Here is a rebase of this experimental patch.  I think the basic
mechanics are promising, but we haven't agreed on a UX.  I hope we can
figure this out.

Restating the choice made in this branch of the experiment:  Here I
try to be just like DB2 (if I understood its manual correctly).
In DB2, you can use names like "en_US" if you don't care about
changes, and names like "CLDR181_en_US" if you do.  It's the user's
choice to use the second kind to avoid "unexpected effects on
applications or database objects" after upgrades.  Translated to
PostgreSQL concepts, you can use a database default ICU locale like
"en-US" if you don't care and "67:en-US" if you do, and for COLLATION
objects it's the same.  The convention I tried in this patch is that
you use either "en-US-x-icu" (which points to "en-US") or
"en-US-x-icu67" (which points to "67:en-US") depending on whether you
care about this problem.

I recognise that this is a bit cheesy, it's all the user's problem to
deal with or ignore.

An alternative mentioned by Peter E was that the locale names
shouldn't carry the prefix, but somehow we should have a list of ICU
versions to search for a matching datcollversion/collversion.  How
would that look?  Perhaps a GUC, icu_library_versions = '63, 67, 71'?
There is a currently natural and smallish range of supported versions,
probably something like 54 ... U_ICU_VERSION_MAJOR_NUM, but it seems a
bit weird to try to dlopen ~25 libraries or whatever it might be...
Do you think we should try to code this up?

I haven't tried it, but the main usability problem I predict with that
idea is this:  It can cope with a scenario where you created a
database with ICU 63 and started using a default of "en" and maybe
some explicit fr-x-icu or whatever, and then you upgrade to a new
postgres binary using ICU 71, and, as long as you still have ICU 63
installed it'll just magicaly keep using 63, now via dlopen().  But it
doesn't provide a way for me to create a new database that uses 63 on
purpose when I know what I'm doing.  There are various reasons I might
want to do that.

Maybe the ideas could be combined?  Perhaps "en" means "create using
binary's linked ICU, open using search-by-collversion", while "67:en"
explicitly says which to use?

Changes since last version:

 * Now it just uses the default dlopen() search path, unless you set
icu_library_path.  Is that a security problem?  It's pretty
convenient, because it means you can just "apt-get install libicu63"
(or local equivalent) and that's all, now 63 is available.

 * To try the idea out, I made it automatically create "*-x-icu67"
alongside the regular "-x-icu" collation objects at initdb time.


Attachments:

  [application/x-patch] v5-0001-WIP-Multi-version-ICU.patch (30.9K, ../../CA+hUKGL36vXMfcaDq+U1ZkoSsdfFnNx7GxhGM7aYzEbKs1W0=Q@mail.gmail.com/2-v5-0001-WIP-Multi-version-ICU.patch)
  download | inline diff:
From d3e83d0aa5cbb3eb192a2f66d68623cd3b1595b4 Mon Sep 17 00:00:00 2001
From: Thomas Munro <[email protected]>
Date: Wed, 8 Jun 2022 17:43:53 +1200
Subject: [PATCH v5] WIP: Multi-version ICU.

Add a layer of indirection when accessing ICU, so that multiple major
versions of the library can be used at once.  Versions other than the
one that PostgreSQL was linked against are opened with dlopen(), but we
refuse to open version higher than the one were were compiled against.
The ABI might change in future releases so that wouldn't be safe.

By default, the system linker's default search path is used to find
libraries, but icu_library_path may be used to specify an absolute path
to look in.  ICU libraries are expected to have been built without ICU's
--disable-renaming option.  That is, major versions must use distinct
symbol names.

This arrangement means that at least one major version of ICU is always
available -- the one that PostgreSQL was linked again.  It should be
simple on most software distributions to install extra versions using a
package manager, or to build extra libraries as required, to access
older ICU releases.  For example, on Debian bullseye the packages are
named libicu63, libicu67, libicu71.

In this version of the patch, '63:en' used as a database default locale
or COLLATION object requests ICU library 63, and 'en' requests the
library that is linked against the postgres executable.

XXX Many other designs possible, to discuss!

Discussion: https://postgr.es/m/CA%2BhUKGL4VZRpP3CkjYQkv4RQ6pRYkPkSNgKSxFBwciECQ0mEuQ%40mail.gmail.com
---
 src/backend/access/hash/hashfunc.c   |  16 +-
 src/backend/commands/collationcmds.c |  20 ++
 src/backend/utils/adt/formatting.c   |  53 +++-
 src/backend/utils/adt/pg_locale.c    | 364 ++++++++++++++++++++++++++-
 src/backend/utils/adt/varchar.c      |  16 +-
 src/backend/utils/adt/varlena.c      |  56 +++--
 src/backend/utils/misc/guc_tables.c  |  14 ++
 src/include/utils/pg_locale.h        |  73 ++++++
 src/tools/pgindent/typedefs.list     |   3 +
 9 files changed, 549 insertions(+), 66 deletions(-)

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index b57ed946c4..0a61538efd 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -298,11 +298,11 @@ hashtext(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -355,11 +355,11 @@ hashtextextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index fcfc02d2ae..26e747d9d7 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -812,6 +812,26 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 					CreateComments(collid, CollationRelationId, 0,
 								   icucomment);
 			}
+
+			/* Also create an object pinned to an ICU major version. */
+			collid = CollationCreate(psprintf("%s-x-icu-%d", langtag, U_ICU_VERSION_MAJOR_NUM),
+									 nspid, GetUserId(),
+									 COLLPROVIDER_ICU, true, -1,
+									 NULL, NULL,
+									 psprintf("%d:%s", U_ICU_VERSION_MAJOR_NUM, iculocstr),
+									 get_collation_actual_version(COLLPROVIDER_ICU, iculocstr),
+									 true, true);
+			if (OidIsValid(collid))
+			{
+				ncreated++;
+
+				CommandCounterIncrement();
+
+				icucomment = get_icu_locale_comment(name);
+				if (icucomment)
+					CreateComments(collid, CollationRelationId, 0,
+								   icucomment);
+			}
 		}
 	}
 #endif							/* USE_ICU */
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index 26f498b5df..0c3c7724d7 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -1599,6 +1599,11 @@ typedef int32_t (*ICU_Convert_Func) (UChar *dest, int32_t destCapacity,
 									 const UChar *src, int32_t srcLength,
 									 const char *locale,
 									 UErrorCode *pErrorCode);
+typedef int32_t (*ICU_Convert_BI_Func) (UChar *dest, int32_t destCapacity,
+										const UChar *src, int32_t srcLength,
+										UBreakIterator *bi,
+										const char *locale,
+										UErrorCode *pErrorCode);
 
 static int32_t
 icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
@@ -1623,18 +1628,41 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("case conversion failed: %s", u_errorName(status))));
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
 	return len_dest;
 }
 
+/*
+ * Like icu_convert_case, but func takes a break iterator (which we don't
+ * make use of).
+ */
 static int32_t
-u_strToTitle_default_BI(UChar *dest, int32_t destCapacity,
-						const UChar *src, int32_t srcLength,
-						const char *locale,
-						UErrorCode *pErrorCode)
+icu_convert_case_bi(ICU_Convert_BI_Func func, pg_locale_t mylocale,
+					UChar **buff_dest, UChar *buff_source, int32_t len_source)
 {
-	return u_strToTitle(dest, destCapacity, src, srcLength,
-						NULL, locale, pErrorCode);
+	UErrorCode	status;
+	int32_t		len_dest;
+
+	len_dest = len_source;		/* try first with same length */
+	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+	status = U_ZERO_ERROR;
+	len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+					mylocale->info.icu.locale, &status);
+	if (status == U_BUFFER_OVERFLOW_ERROR)
+	{
+		/* try again with adjusted length */
+		pfree(*buff_dest);
+		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+		status = U_ZERO_ERROR;
+		len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+						mylocale->info.icu.locale, &status);
+	}
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
+	return len_dest;
 }
 
 #endif							/* USE_ICU */
@@ -1702,7 +1730,8 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToLower, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToLower,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1824,7 +1853,8 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToUpper, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToUpper,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1947,8 +1977,9 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToTitle_default_BI, mylocale,
-										&buff_conv, buff_uchar, len_uchar);
+			len_conv = icu_convert_case_bi(PG_ICU_LIB(mylocale)->strToTitle,
+										   mylocale,
+										   &buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 2b42d9ccd8..bf76516406 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -58,6 +58,7 @@
 #include "catalog/pg_collation.h"
 #include "catalog/pg_control.h"
 #include "mb/pg_wchar.h"
+#include "miscadmin.h"
 #include "utils/builtins.h"
 #include "utils/formatting.h"
 #include "utils/guc_hooks.h"
@@ -69,6 +70,7 @@
 
 #ifdef USE_ICU
 #include <unicode/ucnv.h>
+#include <unicode/ustring.h>
 #endif
 
 #ifdef __GLIBC__
@@ -79,14 +81,31 @@
 #include <shlwapi.h>
 #endif
 
+#include <dlfcn.h>
+
 #define		MAX_L10N_DATA		80
 
+#ifdef USE_ICU
+
+/*
+ * We don't want to call into dlopen'd ICU libraries that are newer than the
+ * one we were compiled and linked against, just in case there is an
+ * incompatible API change.
+ */
+#define PG_MAX_ICU_MAJOR_VERSION U_ICU_VERSION_MAJOR_NUM
+
+/* An old ICU release that we know has the right API. */
+#define PG_MIN_ICU_MAJOR_VERSION 54
+
+#endif
+
 
 /* GUC settings */
 char	   *locale_messages;
 char	   *locale_monetary;
 char	   *locale_numeric;
 char	   *locale_time;
+char	   *icu_library_path;
 
 /*
  * lc_time localization cache.
@@ -1398,29 +1417,343 @@ lc_ctype_is_c(Oid collation)
 	return (lookup_collation_cache(collation, true))->ctype_is_c;
 }
 
+#ifdef USE_ICU
+
 struct pg_locale_struct default_locale;
 
+/* Linked list of ICU libraries we have loaded. */
+static pg_icu_library *icu_library_list = NULL;
+
+/*
+ * Free an ICU library.  pg_icu_library objects that are successfully
+ * constructed stick around for the lifetime of the backend, but this is used
+ * to clean up if initialization fails.
+ */
+static void
+free_icu_library(pg_icu_library *lib)
+{
+	if (lib->libicui18n_handle)
+		dlclose(lib->libicui18n_handle);
+	if (lib->libicuuc_handle)
+		dlclose(lib->libicuuc_handle);
+	pfree(lib);
+}
+
+static void *
+get_icu_function(void *handle, const char *function, int version)
+{
+	char		name[80];
+
+	snprintf(name, sizeof(name), "%s_%d", function, version);
+
+	return dlsym(handle, name);
+}
+
+/*
+ * Probe a dynamically loaded library to see which major version of ICU it
+ * contains.
+ */
+static int
+get_icu_library_major_version(void *handle)
+{
+	for (int i = PG_MIN_ICU_MAJOR_VERSION; i <= PG_MAX_ICU_MAJOR_VERSION; ++i)
+		if (get_icu_function(handle, "ucol_open", i) ||
+			get_icu_function(handle, "u_strToUpper", i))
+			return i;
+
+	/*
+	 * It's a later version we don't dare use, an old version we don't
+	 * support, an ICU build with symbol suffixes disabled, or not ICU.
+	 */
+	return -1;
+}
+
+/*
+ * We have to load a couple of different libraries, so we'll reuse the code to
+ * do that.
+ */
+static void *
+load_icu_library(pg_icu_library *lib, const char *name)
+{
+	void	   *handle;
+	int			found_major_version;
+
+	handle = dlopen(name, RTLD_NOW | RTLD_GLOBAL);
+	if (handle == NULL)
+	{
+		int			errno_save = errno;
+
+		free_icu_library(lib);
+		errno = errno_save;
+
+		ereport(ERROR,
+				(errmsg("could not load library \"%s\": %m", name)));
+	}
+
+	found_major_version = get_icu_library_major_version(handle);
+	if (found_major_version < 0)
+	{
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("could not find compatible ICU major version in library \"%s\"",
+						name)));
+	}
+
+	if (found_major_version != lib->major_version)
+	{
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("expected to find ICU major version %d in library \"%s\", but found %d",
+						lib->major_version, name, found_major_version)));
+	}
+
+	return handle;
+}
+
+/*
+ * Given an ICU major version number, return the object we need to access it,
+ * or fail while trying to load it.
+ */
+static pg_icu_library *
+get_icu_library(int major_version)
+{
+	pg_icu_library *lib;
+
+	Assert(major_version >= PG_MIN_ICU_MAJOR_VERSION &&
+		   major_version <= PG_MAX_ICU_MAJOR_VERSION);
+
+	/* Try to find it in our list of existing libraries. */
+	for (lib = icu_library_list; lib; lib = lib->next)
+		if (lib->major_version == major_version)
+			return lib;
+
+	/* Make a new entry. */
+	lib = MemoryContextAllocZero(TopMemoryContext, sizeof(*lib));
+	if (major_version == U_ICU_VERSION_MAJOR_NUM)
+	{
+		/*
+		 * This is the version we were compiled and linked against.  Simply
+		 * assign the function pointers.
+		 *
+		 * These assignments will fail to compile if an incompatible API
+		 * change is made to some future version of ICU, at which point we
+		 * might need to consider special treatment for different major
+		 * version ranges, with intermediate trampoline functions.
+		 */
+		lib->major_version = major_version;
+		lib->open = ucol_open;
+		lib->close = ucol_close;
+		lib->getVersion = ucol_getVersion;
+		lib->versionToString = u_versionToString;
+		lib->strcoll = ucol_strcoll;
+		lib->strcollUTF8 = ucol_strcollUTF8;
+		lib->getSortKey = ucol_getSortKey;
+		lib->nextSortKeyPart = ucol_nextSortKeyPart;
+		lib->setUTF8 = uiter_setUTF8;
+		lib->errorName = u_errorName;
+		lib->strToUpper = u_strToUpper;
+		lib->strToLower = u_strToLower;
+		lib->strToTitle = u_strToTitle;
+
+		/*
+		 * Also assert the size of a couple of types used as output buffers,
+		 * as a canary to tell us to add extra padding in the (unlikely) event
+		 * that a later release makes these values smaller.
+		 */
+		StaticAssertStmt(U_MAX_VERSION_STRING_LENGTH == 20,
+						 "u_versionToString output buffer size changed incompatibly");
+		StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
+						 "ucol_getVersion output buffer size changed incompatibly");
+	}
+	else
+	{
+		/* This is an older version, so we'll need to use dlopen(). */
+		char		libicui18n_name[MAXPGPATH];
+		char		libicuuc_name[MAXPGPATH];
+
+		/*
+		 * We don't like to open versions newer than what we're linked
+		 * against, to reduce the risk of an API change biting us.
+		 */
+		if (major_version > U_ICU_VERSION_MAJOR_NUM)
+			elog(ERROR, "ICU major version %d higher than linked version %d, refusing to open",
+				 major_version, U_ICU_VERSION_MAJOR_NUM);
+
+		lib->major_version = major_version;
+
+		/*
+		 * See
+		 * https://unicode-org.github.io/icu/userguide/icu4c/packaging.html#icu-versions
+		 * for conventions on library naming on POSIX and Windows systems.
+		 */
+
+		/* Load the collation library. */
+		snprintf(libicui18n_name,
+				 sizeof(libicui18n_name),
+#ifdef WIN32
+				 "%s%sicui18n%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#else
+				 "%s%slibicui18n" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicui18n_handle = load_icu_library(lib, libicui18n_name);
+
+		/* Load the ctype library. */
+		snprintf(libicuuc_name,
+				 sizeof(libicuuc_name),
+#ifdef WIN32
+				 "%s%sicuuc%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#else
+				 "%s%slibicuuc" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicuuc_handle = load_icu_library(lib, libicuuc_name);
+
+		/* Look up all the functions we need. */
+		lib->open = get_icu_function(lib->libicui18n_handle,
+									 "ucol_open",
+									 major_version);
+		lib->close = get_icu_function(lib->libicui18n_handle,
+									  "ucol_close",
+									  major_version);
+		lib->getVersion = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getVersion",
+										   major_version);
+		lib->versionToString = get_icu_function(lib->libicui18n_handle,
+												"u_versionToString",
+												major_version);
+		lib->strcoll = get_icu_function(lib->libicui18n_handle,
+										"ucol_strcoll",
+										major_version);
+		lib->strcollUTF8 = get_icu_function(lib->libicui18n_handle,
+											"ucol_strcollUTF8",
+											major_version);
+		lib->getSortKey = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getSortKey",
+										   major_version);
+		lib->nextSortKeyPart = get_icu_function(lib->libicui18n_handle,
+												"ucol_nextSortKeyPart",
+												major_version);
+		lib->setUTF8 = get_icu_function(lib->libicui18n_handle,
+										"uiter_setUTF8",
+										major_version);
+		lib->errorName = get_icu_function(lib->libicui18n_handle,
+										  "u_errorName",
+										  major_version);
+		lib->strToUpper = get_icu_function(lib->libicuuc_handle,
+										   "u_strToUpper",
+										   major_version);
+		lib->strToLower = get_icu_function(lib->libicuuc_handle,
+										   "u_strToLower",
+										   major_version);
+		lib->strToTitle = get_icu_function(lib->libicuuc_handle,
+										   "u_strToTitle",
+										   major_version);
+		if (!lib->open ||
+			!lib->close ||
+			!lib->getVersion ||
+			!lib->versionToString ||
+			!lib->strcoll ||
+			!lib->strcollUTF8 ||
+			!lib->getSortKey ||
+			!lib->nextSortKeyPart ||
+			!lib->setUTF8 ||
+			!lib->errorName ||
+			!lib->strToUpper ||
+			!lib->strToLower ||
+			!lib->strToTitle)
+		{
+			free_icu_library(lib);
+			ereport(ERROR,
+					(errmsg("could not find expected symbols in library \"%s\"",
+							libicui18n_name)));
+		}
+	}
+
+	lib->next = icu_library_list;
+	icu_library_list = lib;
+
+	return lib;
+}
+
+/*
+ * Look up the library to use for a given collcollate string.
+ */
+static pg_icu_library *
+get_icu_library_for_collation(const char *collcollate, const char **rest)
+{
+	int			major_version;
+	char	   *separator;
+	char	   *after_prefix;
+
+	separator = strchr(collcollate, ':');
+
+	/*
+	 * If it's a traditional value without a prefix, use the library we are
+	 * linked against.
+	 */
+	if (separator == NULL)
+	{
+		*rest = collcollate;
+		return get_icu_library(U_ICU_VERSION_MAJOR_NUM);
+	}
+
+	/* If it has a prefix, interpret it as an ICU major version. */
+	major_version = strtol(collcollate, &after_prefix, 10);
+	if (after_prefix != separator)
+		elog(ERROR,
+			 "could not parse ICU major library version: \"%s\"",
+			 collcollate);
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+		elog(ERROR,
+			 "ICU major library verision out of supported range: \"%s\"",
+			 collcollate);
+
+	/* The part after the separate will be passed to the library. */
+	*rest = separator + 1;
+
+	return get_icu_library(major_version);
+}
+
+#endif
+
 void
 make_icu_collator(const char *iculocstr,
 				  struct pg_locale_struct *resultp)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(iculocstr, &iculocstr);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(iculocstr, &status);
+	collator = lib->open(iculocstr, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						iculocstr, u_errorName(status))));
+						iculocstr, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	/*
+	 * XXX can we just drop this cruft and make 54 the minimum supported
+	 * version?
+	 */
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, iculocstr);
 
 	/* We will leak this string if the caller errors later :-( */
 	resultp->info.icu.locale = MemoryContextStrdup(TopMemoryContext, iculocstr);
 	resultp->info.icu.ucol = collator;
+	resultp->info.icu.lib = lib;
 #else							/* not USE_ICU */
 	/* could get here if a collation was created by a build with ICU */
 	ereport(ERROR,
@@ -1651,21 +1984,23 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 #ifdef USE_ICU
 	if (collprovider == COLLPROVIDER_ICU)
 	{
+		pg_icu_library *lib;
 		UCollator  *collator;
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
 
+		lib = get_icu_library_for_collation(collcollate, &collcollate);
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collcollate, &status);
+		collator = lib->open(collcollate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collcollate, u_errorName(status))));
-		ucol_getVersion(collator, versioninfo);
-		ucol_close(collator);
+							collcollate, lib->errorName(status))));
+		lib->getVersion(collator, versioninfo);
+		lib->close(collator);
 
-		u_versionToString(versioninfo, buf);
+		lib->versionToString(versioninfo, buf);
 		collversion = pstrdup(buf);
 	}
 	else
@@ -1733,6 +2068,8 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 
 
 #ifdef USE_ICU
+
+
 /*
  * Converter object for converting between ICU's UChar strings and C strings
  * in database encoding.  Since the database encoding doesn't change, we only
@@ -1954,19 +2291,22 @@ void
 check_icu_locale(const char *icu_locale)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(icu_locale, &icu_locale);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(icu_locale, &status);
+	collator = lib->open(icu_locale, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						icu_locale, u_errorName(status))));
+						icu_locale, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	/* XXX can we just drop this cruft? */
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, icu_locale);
-	ucol_close(collator);
+	lib->close(collator);
 #else
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index 68e2e6f7a7..e0c86870e0 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -1026,11 +1026,11 @@ hashbpchar(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, keydata, keylen);
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -1087,11 +1087,11 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c5e7ee7ca2..cf891a5654 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1667,13 +1667,14 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					UErrorCode	status;
 
 					status = U_ZERO_ERROR;
-					result = ucol_strcollUTF8(mylocale->info.icu.ucol,
-											  arg1, len1,
-											  arg2, len2,
-											  &status);
+					result = PG_ICU_LIB(mylocale)->strcollUTF8(PG_ICU_COL(mylocale),
+															   arg1, len1,
+															   arg2, len2,
+															   &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
-								(errmsg("collation failed: %s", u_errorName(status))));
+								(errmsg("collation failed: %s",
+										PG_ICU_LIB(mylocale)->errorName(status))));
 				}
 				else
 #endif
@@ -1686,9 +1687,9 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					ulen1 = icu_to_uchar(&uchar1, arg1, len1);
 					ulen2 = icu_to_uchar(&uchar2, arg2, len2);
 
-					result = ucol_strcoll(mylocale->info.icu.ucol,
-										  uchar1, ulen1,
-										  uchar2, ulen2);
+					result = PG_ICU_LIB(mylocale)->strcoll(PG_ICU_COL(mylocale),
+														   uchar1, ulen1,
+														   uchar2, ulen2);
 
 					pfree(uchar1);
 					pfree(uchar2);
@@ -2388,13 +2389,14 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				UErrorCode	status;
 
 				status = U_ZERO_ERROR;
-				result = ucol_strcollUTF8(sss->locale->info.icu.ucol,
-										  a1p, len1,
-										  a2p, len2,
-										  &status);
+				result = PG_ICU_LIB(sss->locale)->strcollUTF8(PG_ICU_COL(sss->locale),
+															  a1p, len1,
+															  a2p, len2,
+															  &status);
 				if (U_FAILURE(status))
 					ereport(ERROR,
-							(errmsg("collation failed: %s", u_errorName(status))));
+							(errmsg("collation failed: %s",
+									PG_ICU_LIB(sss->locale)->errorName(status))));
 			}
 			else
 #endif
@@ -2407,9 +2409,9 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				ulen1 = icu_to_uchar(&uchar1, a1p, len1);
 				ulen2 = icu_to_uchar(&uchar2, a2p, len2);
 
-				result = ucol_strcoll(sss->locale->info.icu.ucol,
-									  uchar1, ulen1,
-									  uchar2, ulen2);
+				result = PG_ICU_LIB(sss->locale)->strcoll(PG_ICU_COL(sss->locale),
+														  uchar1, ulen1,
+														  uchar2, ulen2);
 
 				pfree(uchar1);
 				pfree(uchar2);
@@ -2569,24 +2571,24 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 					uint32_t	state[2];
 					UErrorCode	status;
 
-					uiter_setUTF8(&iter, sss->buf1, len);
+					PG_ICU_LIB(sss->locale)->setUTF8(&iter, sss->buf1, len);
 					state[0] = state[1] = 0;	/* won't need that again */
 					status = U_ZERO_ERROR;
-					bsize = ucol_nextSortKeyPart(sss->locale->info.icu.ucol,
-												 &iter,
-												 state,
-												 (uint8_t *) sss->buf2,
-												 Min(sizeof(Datum), sss->buflen2),
-												 &status);
+					bsize = PG_ICU_LIB(sss->locale)->nextSortKeyPart(PG_ICU_COL(sss->locale),
+																	 &iter,
+																	 state,
+																	 (uint8_t *) sss->buf2,
+																	 Min(sizeof(Datum), sss->buflen2),
+																	 &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
 								(errmsg("sort key generation failed: %s",
-										u_errorName(status))));
+										PG_ICU_LIB(sss->locale)->errorName(status))));
 				}
 				else
-					bsize = ucol_getSortKey(sss->locale->info.icu.ucol,
-											uchar, ulen,
-											(uint8_t *) sss->buf2, sss->buflen2);
+					bsize = PG_ICU_LIB(sss->locale)->getSortKey(PG_ICU_COL(sss->locale),
+																uchar, ulen,
+																(uint8_t *) sss->buf2, sss->buflen2);
 			}
 			else
 #endif
diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c
index 05ab087934..e60081c384 100644
--- a/src/backend/utils/misc/guc_tables.c
+++ b/src/backend/utils/misc/guc_tables.c
@@ -3922,6 +3922,20 @@ struct config_string ConfigureNamesString[] =
 		NULL, NULL, NULL
 	},
 
+	{
+		{"icu_library_path", PGC_SUSET, CLIENT_CONN_OTHER,
+			gettext_noop("Sets the path for dynamically loadable ICU libraries."),
+			gettext_noop("If versions of ICU other than the one that "
+						 "PostgreSQL is linked against, they will be open "
+						 "from this path.  If empty, the system linker search "
+						 "path will be used."),
+			GUC_SUPERUSER_ONLY
+		},
+		&icu_library_path,
+		"",
+		NULL, NULL, NULL
+	},
+
 	{
 		{"krb_server_keyfile", PGC_SIGHUP, CONN_AUTH_AUTH,
 			gettext_noop("Sets the location of the Kerberos server key file."),
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index a875942123..59613e4f56 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -17,6 +17,7 @@
 #endif
 #ifdef USE_ICU
 #include <unicode/ucol.h>
+#include <unicode/ubrk.h>
 #endif
 
 #ifdef USE_ICU
@@ -40,6 +41,7 @@ extern PGDLLIMPORT char *locale_messages;
 extern PGDLLIMPORT char *locale_monetary;
 extern PGDLLIMPORT char *locale_numeric;
 extern PGDLLIMPORT char *locale_time;
+extern PGDLLIMPORT char *icu_library_path;
 
 /* lc_time localization cache */
 extern PGDLLIMPORT char *localized_abbrev_days[];
@@ -63,6 +65,71 @@ extern struct lconv *PGLC_localeconv(void);
 
 extern void cache_locale_time(void);
 
+#ifdef USE_ICU
+
+/*
+ * An ICU library version that we're either linked against or have loaded at
+ * runtime.
+ */
+typedef struct pg_icu_library
+{
+	int			major_version;
+	void	   *libicui18n_handle;
+	void	   *libicuuc_handle;
+	UCollator  *(*open) (const char *loc, UErrorCode *status);
+	void		(*close) (UCollator *coll);
+	void		(*getVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*versionToString) (const UVersionInfo versionArray,
+									char *versionString);
+				UCollationResult(*strcoll) (const UCollator *coll,
+											const UChar *source,
+											int32_t sourceLength,
+											const UChar *target,
+											int32_t targetLength);
+				UCollationResult(*strcollUTF8) (const UCollator *coll,
+												const char *source,
+												int32_t sourceLength,
+												const char *target,
+												int32_t targetLength,
+												UErrorCode *status);
+	int32_t		(*getSortKey) (const UCollator *coll,
+							   const UChar *source,
+							   int32_t sourceLength,
+							   uint8_t *result,
+							   int32_t resultLength);
+	int32_t		(*nextSortKeyPart) (const UCollator *coll,
+									UCharIterator *iter,
+									uint32_t state[2],
+									uint8_t *dest,
+									int32_t count,
+									UErrorCode *status);
+	void		(*setUTF8) (UCharIterator *iter,
+							const char *s,
+							int32_t length);
+	const char *(*errorName) (UErrorCode code);
+	int32_t		(*strToUpper) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToLower) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToTitle) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UBreakIterator *titleIter,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	struct pg_icu_library *next;
+} pg_icu_library;
+
+#endif
 
 /*
  * We define our own wrapper around locale_t so we can keep the same
@@ -84,12 +151,18 @@ struct pg_locale_struct
 		{
 			const char *locale;
 			UCollator  *ucol;
+			pg_icu_library *lib;
 		}			icu;
 #endif
 		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
 	}			info;
 };
 
+#ifdef USE_ICU
+#define PG_ICU_LIB(x) ((x)->info.icu.lib)
+#define PG_ICU_COL(x) ((x)->info.icu.ucol)
+#endif
+
 typedef struct pg_locale_struct *pg_locale_t;
 
 extern PGDLLIMPORT struct pg_locale_struct default_locale;
diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list
index d9b839c979..0ccbbb711e 100644
--- a/src/tools/pgindent/typedefs.list
+++ b/src/tools/pgindent/typedefs.list
@@ -1101,6 +1101,7 @@ HeapTupleTableSlot
 HistControl
 HotStandbyState
 I32
+ICU_Convert_BI_Func
 ICU_Convert_Func
 ID
 INFIX
@@ -2854,6 +2855,7 @@ TypeName
 U
 U32
 U8
+UBreakIterator
 UChar
 UCharIterator
 UColAttribute
@@ -3482,6 +3484,7 @@ pg_funcptr_t
 pg_gssinfo
 pg_hmac_ctx
 pg_hmac_errno
+pg_icu_library
 pg_int64
 pg_local_to_utf_combined
 pg_locale_t
-- 
2.30.2



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-10-22 01:22  Thomas Munro <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 3 replies; 57+ messages in thread

From: Thomas Munro @ 2022-10-22 01:22 UTC (permalink / raw)
  To: Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sat, Oct 22, 2022 at 10:24 AM Thomas Munro <[email protected]> wrote:
> ... But it
> doesn't provide a way for me to create a new database that uses 63 on
> purpose when I know what I'm doing.  There are various reasons I might
> want to do that.

Thinking some more about this, I guess that could be addressed by
having an explicit way to request either the library version or
collversion-style version when creating a database or collation, but
not actually storing it in daticulocale/colliculocale.  That could be
done either as part of the string that is trimmed off before storing
it (so it's only used briefly during creation to find a non-default
library)... Perhaps that'd look like initdb --icu-locale "67:en" (ICU
library version) or "154.14:en" (individual collation version) or some
new syntax in a few places.  Thereafter, it would always be looked up
by searching for the right library by [dat]collversion as Peter E
suggested.

Let me try harder to vocalise some more thoughts that have stopped me
from trying to code the search-by-collversion design so far:

Suppose your pgdata encounters a PostgreSQL linked against a later ICU
library, most likely after an OS upgrade or migratoin, a pg_upgrade,
or via streaming replication.  You might get a new error "can't find
ICU collation 'en' with version '153.14'; HINT: install missing ICU
library version", and somehow you'll have to work out which one might
contain 'en' v153.14 and install it with apt-get etc.  Then it'll
magically work: your postgres linked against (say) 71 will happily
work with the dlopen'd 67.  This is enough if you want to stay on 67
until the heat death of the universe.  So far so good.

Problem 1:  Suppose you're ready to start using (say) v72.  I guess
you'd use the REFRESH command, which would open the main linked ICU's
collversion and stamp that into the catalogue, at which point new
sessions would start using that, and then you'd have to rebuild all
your indexes (with no help from PG to tell you how to find everything
that needs to be rebuilt, as belaboured in previous reverted work).
Aside from the possibility of getting the rebuilding job wrong (as
belaboured elsewhere), it's not great, because there is still a
transitional period where you can be using the wrong version for your
data.  So this requires some careful planning and understanding from
the administrator.

I admit that the upgrade story is a tiny bit better than the v5
DB2-style patch, which starts using the new version immediately if you
didn't use a prefix (and logs the usual warnings about collversion
mismatch) instead of waiting for you to run REFRESH.  But both of them
have a phase where they might use the wrong library to access an
index.  That's dissatisfying, and leads me to prefer the simple
DB2-style solution that at least admits up front that it's not very
clever.  The DB2-style patch could be improved a bit here with the
addition of one more GUC: default_icu_library, so the administrator,
rather than the packager, remains in control of which version we use
for non-prefixed iculocale values (likely to be what almost everyone
is interested in), defaulting to what the packager linked against.
I've added that to the patch for illustration (though obviously the
error messages produced by collversion mismatch could use some
adjustment, ie to clarify that the warning might be cleared by
installing and selecting a different library version).

Problem 2:  If ICU 67 ever decides to report a different version for a
given collation (would it ever do that?  I don't expect so, but ...),
we'd be unable to open the collation with the search-by-collversion
design, and potentially the database.  What is a user supposed to do
then?  Presumably our error/hint for that would be "please insert the
correct ICU library into drive A", but now there is no correct
library; if you can even diagnose what's happened, I guess you might
downgrade the ICU library using package tools or whatever if possible,
but otherwise you'd be stuck, if you just can't get the right library.
Is this a problem?  Would you want to be able to say "I don't care,
computer, please just press on"?  So I think we need a way to turn off
the search-by-collversion thing.  How should it look?

I'd love to hear others' thoughts on how we can turn this into a
workable solution.  Hopefully while staying simple...


Attachments:

  [text/x-patch] v6-0001-WIP-Multi-version-ICU.patch (32.7K, ../../CA+hUKGKq=iLH3bY+nK7v8b2zBCuKOk-fe0cP0it2RxNaWFVxYA@mail.gmail.com/2-v6-0001-WIP-Multi-version-ICU.patch)
  download | inline diff:
From 0355984c9a80ff15bfac51677fea30b9be68226b Mon Sep 17 00:00:00 2001
From: Thomas Munro <[email protected]>
Date: Wed, 8 Jun 2022 17:43:53 +1200
Subject: [PATCH v6] WIP: Multi-version ICU.

Add a layer of indirection when accessing ICU, so that multiple major
versions of the library can be used at once.  Versions other than the
one that PostgreSQL was linked against are opened with dlopen(), but we
refuse to open version higher than the one were were compiled against.
The ABI might change in future releases so that wouldn't be safe.

By default, the system linker's default search path is used to find
libraries, but icu_library_path may be used to specify an absolute path
to look in.  ICU libraries are expected to have been built without ICU's
--disable-renaming option.  That is, major versions must use distinct
symbol names.

This arrangement means that at least one major version of ICU is always
available -- the one that PostgreSQL was linked again.  It should be
simple on most software distributions to install extra versions using a
package manager, or to build extra libraries as required, to access
older ICU releases.  For example, on Debian bullseye the packages are
named libicu63, libicu67, libicu71.

In this version of the patch, '63:en' used as a database default locale
or COLLATION object requests ICU library 63, and 'en' requests the
library version seleted by the GUC default_icu_library_version,
defaulting to the version that the executable is linked against.

XXX Many other designs possible, to discuss!

Discussion: https://postgr.es/m/CA%2BhUKGL4VZRpP3CkjYQkv4RQ6pRYkPkSNgKSxFBwciECQ0mEuQ%40mail.gmail.com
---
 src/backend/access/hash/hashfunc.c            |  16 +-
 src/backend/commands/collationcmds.c          |  20 +
 src/backend/utils/adt/formatting.c            |  53 ++-
 src/backend/utils/adt/pg_locale.c             | 376 +++++++++++++++++-
 src/backend/utils/adt/varchar.c               |  16 +-
 src/backend/utils/adt/varlena.c               |  56 +--
 src/backend/utils/misc/guc_tables.c           |  28 ++
 src/backend/utils/misc/postgresql.conf.sample |   5 +
 src/include/utils/pg_locale.h                 |  74 ++++
 src/tools/pgindent/typedefs.list              |   3 +
 10 files changed, 581 insertions(+), 66 deletions(-)

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index b57ed946c4..0a61538efd 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -298,11 +298,11 @@ hashtext(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -355,11 +355,11 @@ hashtextextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index fcfc02d2ae..26e747d9d7 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -812,6 +812,26 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 					CreateComments(collid, CollationRelationId, 0,
 								   icucomment);
 			}
+
+			/* Also create an object pinned to an ICU major version. */
+			collid = CollationCreate(psprintf("%s-x-icu-%d", langtag, U_ICU_VERSION_MAJOR_NUM),
+									 nspid, GetUserId(),
+									 COLLPROVIDER_ICU, true, -1,
+									 NULL, NULL,
+									 psprintf("%d:%s", U_ICU_VERSION_MAJOR_NUM, iculocstr),
+									 get_collation_actual_version(COLLPROVIDER_ICU, iculocstr),
+									 true, true);
+			if (OidIsValid(collid))
+			{
+				ncreated++;
+
+				CommandCounterIncrement();
+
+				icucomment = get_icu_locale_comment(name);
+				if (icucomment)
+					CreateComments(collid, CollationRelationId, 0,
+								   icucomment);
+			}
 		}
 	}
 #endif							/* USE_ICU */
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index 26f498b5df..0c3c7724d7 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -1599,6 +1599,11 @@ typedef int32_t (*ICU_Convert_Func) (UChar *dest, int32_t destCapacity,
 									 const UChar *src, int32_t srcLength,
 									 const char *locale,
 									 UErrorCode *pErrorCode);
+typedef int32_t (*ICU_Convert_BI_Func) (UChar *dest, int32_t destCapacity,
+										const UChar *src, int32_t srcLength,
+										UBreakIterator *bi,
+										const char *locale,
+										UErrorCode *pErrorCode);
 
 static int32_t
 icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
@@ -1623,18 +1628,41 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("case conversion failed: %s", u_errorName(status))));
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
 	return len_dest;
 }
 
+/*
+ * Like icu_convert_case, but func takes a break iterator (which we don't
+ * make use of).
+ */
 static int32_t
-u_strToTitle_default_BI(UChar *dest, int32_t destCapacity,
-						const UChar *src, int32_t srcLength,
-						const char *locale,
-						UErrorCode *pErrorCode)
+icu_convert_case_bi(ICU_Convert_BI_Func func, pg_locale_t mylocale,
+					UChar **buff_dest, UChar *buff_source, int32_t len_source)
 {
-	return u_strToTitle(dest, destCapacity, src, srcLength,
-						NULL, locale, pErrorCode);
+	UErrorCode	status;
+	int32_t		len_dest;
+
+	len_dest = len_source;		/* try first with same length */
+	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+	status = U_ZERO_ERROR;
+	len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+					mylocale->info.icu.locale, &status);
+	if (status == U_BUFFER_OVERFLOW_ERROR)
+	{
+		/* try again with adjusted length */
+		pfree(*buff_dest);
+		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+		status = U_ZERO_ERROR;
+		len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+						mylocale->info.icu.locale, &status);
+	}
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
+	return len_dest;
 }
 
 #endif							/* USE_ICU */
@@ -1702,7 +1730,8 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToLower, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToLower,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1824,7 +1853,8 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToUpper, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToUpper,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1947,8 +1977,9 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToTitle_default_BI, mylocale,
-										&buff_conv, buff_uchar, len_uchar);
+			len_conv = icu_convert_case_bi(PG_ICU_LIB(mylocale)->strToTitle,
+										   mylocale,
+										   &buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 2b42d9ccd8..666a79b907 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -58,6 +58,7 @@
 #include "catalog/pg_collation.h"
 #include "catalog/pg_control.h"
 #include "mb/pg_wchar.h"
+#include "miscadmin.h"
 #include "utils/builtins.h"
 #include "utils/formatting.h"
 #include "utils/guc_hooks.h"
@@ -69,6 +70,7 @@
 
 #ifdef USE_ICU
 #include <unicode/ucnv.h>
+#include <unicode/ustring.h>
 #endif
 
 #ifdef __GLIBC__
@@ -79,14 +81,32 @@
 #include <shlwapi.h>
 #endif
 
+#include <dlfcn.h>
+
 #define		MAX_L10N_DATA		80
 
+#ifdef USE_ICU
+
+/*
+ * We don't want to call into dlopen'd ICU libraries that are newer than the
+ * one we were compiled and linked against, just in case there is an
+ * incompatible API change.
+ */
+#define PG_MAX_ICU_MAJOR_VERSION U_ICU_VERSION_MAJOR_NUM
+
+/* An old ICU release that we know has the right API. */
+#define PG_MIN_ICU_MAJOR_VERSION 54
+
+#endif
+
 
 /* GUC settings */
 char	   *locale_messages;
 char	   *locale_monetary;
 char	   *locale_numeric;
 char	   *locale_time;
+char	   *icu_library_path;
+int			default_icu_library_version;
 
 /*
  * lc_time localization cache.
@@ -1398,29 +1418,354 @@ lc_ctype_is_c(Oid collation)
 	return (lookup_collation_cache(collation, true))->ctype_is_c;
 }
 
+#ifdef USE_ICU
+
 struct pg_locale_struct default_locale;
 
+/* Linked list of ICU libraries we have loaded. */
+static pg_icu_library *icu_library_list = NULL;
+
+/*
+ * Free an ICU library.  pg_icu_library objects that are successfully
+ * constructed stick around for the lifetime of the backend, but this is used
+ * to clean up if initialization fails.
+ */
+static void
+free_icu_library(pg_icu_library *lib)
+{
+	if (lib->libicui18n_handle)
+		dlclose(lib->libicui18n_handle);
+	if (lib->libicuuc_handle)
+		dlclose(lib->libicuuc_handle);
+	pfree(lib);
+}
+
+static void *
+get_icu_function(void *handle, const char *function, int version)
+{
+	char		name[80];
+
+	snprintf(name, sizeof(name), "%s_%d", function, version);
+
+	return dlsym(handle, name);
+}
+
+/*
+ * Probe a dynamically loaded library to see which major version of ICU it
+ * contains.
+ */
+static int
+get_icu_library_major_version(void *handle)
+{
+	for (int i = PG_MIN_ICU_MAJOR_VERSION; i <= PG_MAX_ICU_MAJOR_VERSION; ++i)
+		if (get_icu_function(handle, "ucol_open", i) ||
+			get_icu_function(handle, "u_strToUpper", i))
+			return i;
+
+	/*
+	 * It's a later version we don't dare use, an old version we don't
+	 * support, an ICU build with symbol suffixes disabled, or not ICU.
+	 */
+	return -1;
+}
+
+/*
+ * We have to load a couple of different libraries, so we'll reuse the code to
+ * do that.
+ */
+static void *
+load_icu_library(pg_icu_library *lib, const char *name)
+{
+	void	   *handle;
+	int			found_major_version;
+
+	handle = dlopen(name, RTLD_NOW | RTLD_GLOBAL);
+	if (handle == NULL)
+	{
+		int			errno_save = errno;
+
+		free_icu_library(lib);
+		errno = errno_save;
+
+		ereport(ERROR,
+				(errmsg("could not load library \"%s\": %m", name)));
+	}
+
+	found_major_version = get_icu_library_major_version(handle);
+	if (found_major_version < 0)
+	{
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("could not find compatible ICU major version in library \"%s\"",
+						name)));
+	}
+
+	if (found_major_version != lib->major_version)
+	{
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("expected to find ICU major version %d in library \"%s\", but found %d",
+						lib->major_version, name, found_major_version)));
+	}
+
+	return handle;
+}
+
+/*
+ * Given an ICU major version number, return the object we need to access it,
+ * or fail while trying to load it.
+ */
+static pg_icu_library *
+get_icu_library(int major_version)
+{
+	pg_icu_library *lib;
+
+	/* XXX Move range check into guc_table.c? */
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+		elog(ERROR,
+			"ICU version must be between %d and %d",
+			 PG_MIN_ICU_MAJOR_VERSION,
+			 PG_MAX_ICU_MAJOR_VERSION);
+
+	/* Try to find it in our list of existing libraries. */
+	for (lib = icu_library_list; lib; lib = lib->next)
+		if (lib->major_version == major_version)
+			return lib;
+
+	/* Make a new entry. */
+	lib = MemoryContextAllocZero(TopMemoryContext, sizeof(*lib));
+	if (major_version == U_ICU_VERSION_MAJOR_NUM)
+	{
+		/*
+		 * This is the version we were compiled and linked against.  Simply
+		 * assign the function pointers.
+		 *
+		 * These assignments will fail to compile if an incompatible API
+		 * change is made to some future version of ICU, at which point we
+		 * might need to consider special treatment for different major
+		 * version ranges, with intermediate trampoline functions.
+		 */
+		lib->major_version = major_version;
+		lib->open = ucol_open;
+		lib->close = ucol_close;
+		lib->getVersion = ucol_getVersion;
+		lib->versionToString = u_versionToString;
+		lib->strcoll = ucol_strcoll;
+		lib->strcollUTF8 = ucol_strcollUTF8;
+		lib->getSortKey = ucol_getSortKey;
+		lib->nextSortKeyPart = ucol_nextSortKeyPart;
+		lib->setUTF8 = uiter_setUTF8;
+		lib->errorName = u_errorName;
+		lib->strToUpper = u_strToUpper;
+		lib->strToLower = u_strToLower;
+		lib->strToTitle = u_strToTitle;
+
+		/*
+		 * Also assert the size of a couple of types used as output buffers,
+		 * as a canary to tell us to add extra padding in the (unlikely) event
+		 * that a later release makes these values smaller.
+		 */
+		StaticAssertStmt(U_MAX_VERSION_STRING_LENGTH == 20,
+						 "u_versionToString output buffer size changed incompatibly");
+		StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
+						 "ucol_getVersion output buffer size changed incompatibly");
+	}
+	else
+	{
+		/* This is an older version, so we'll need to use dlopen(). */
+		char		libicui18n_name[MAXPGPATH];
+		char		libicuuc_name[MAXPGPATH];
+
+		/*
+		 * We don't like to open versions newer than what we're linked
+		 * against, to reduce the risk of an API change biting us.
+		 */
+		if (major_version > U_ICU_VERSION_MAJOR_NUM)
+			elog(ERROR, "ICU major version %d higher than linked version %d, refusing to open",
+				 major_version, U_ICU_VERSION_MAJOR_NUM);
+
+		lib->major_version = major_version;
+
+		/*
+		 * See
+		 * https://unicode-org.github.io/icu/userguide/icu4c/packaging.html#icu-versions
+		 * for conventions on library naming on POSIX and Windows systems.
+		 */
+
+		/* Load the collation library. */
+		snprintf(libicui18n_name,
+				 sizeof(libicui18n_name),
+#ifdef WIN32
+				 "%s%sicui18n%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#else
+				 "%s%slibicui18n" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicui18n_handle = load_icu_library(lib, libicui18n_name);
+
+		/* Load the ctype library. */
+		snprintf(libicuuc_name,
+				 sizeof(libicuuc_name),
+#ifdef WIN32
+				 "%s%sicuuc%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#else
+				 "%s%slibicuuc" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicuuc_handle = load_icu_library(lib, libicuuc_name);
+
+		/* Look up all the functions we need. */
+		lib->open = get_icu_function(lib->libicui18n_handle,
+									 "ucol_open",
+									 major_version);
+		lib->close = get_icu_function(lib->libicui18n_handle,
+									  "ucol_close",
+									  major_version);
+		lib->getVersion = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getVersion",
+										   major_version);
+		lib->versionToString = get_icu_function(lib->libicui18n_handle,
+												"u_versionToString",
+												major_version);
+		lib->strcoll = get_icu_function(lib->libicui18n_handle,
+										"ucol_strcoll",
+										major_version);
+		lib->strcollUTF8 = get_icu_function(lib->libicui18n_handle,
+											"ucol_strcollUTF8",
+											major_version);
+		lib->getSortKey = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getSortKey",
+										   major_version);
+		lib->nextSortKeyPart = get_icu_function(lib->libicui18n_handle,
+												"ucol_nextSortKeyPart",
+												major_version);
+		lib->setUTF8 = get_icu_function(lib->libicui18n_handle,
+										"uiter_setUTF8",
+										major_version);
+		lib->errorName = get_icu_function(lib->libicui18n_handle,
+										  "u_errorName",
+										  major_version);
+		lib->strToUpper = get_icu_function(lib->libicuuc_handle,
+										   "u_strToUpper",
+										   major_version);
+		lib->strToLower = get_icu_function(lib->libicuuc_handle,
+										   "u_strToLower",
+										   major_version);
+		lib->strToTitle = get_icu_function(lib->libicuuc_handle,
+										   "u_strToTitle",
+										   major_version);
+		if (!lib->open ||
+			!lib->close ||
+			!lib->getVersion ||
+			!lib->versionToString ||
+			!lib->strcoll ||
+			!lib->strcollUTF8 ||
+			!lib->getSortKey ||
+			!lib->nextSortKeyPart ||
+			!lib->setUTF8 ||
+			!lib->errorName ||
+			!lib->strToUpper ||
+			!lib->strToLower ||
+			!lib->strToTitle)
+		{
+			free_icu_library(lib);
+			ereport(ERROR,
+					(errmsg("could not find expected symbols in library \"%s\"",
+							libicui18n_name)));
+		}
+	}
+
+	lib->next = icu_library_list;
+	icu_library_list = lib;
+
+	return lib;
+}
+
+/*
+ * Look up the library to use for a given collcollate string.
+ */
+static pg_icu_library *
+get_icu_library_for_collation(const char *collcollate, const char **rest)
+{
+	int			major_version;
+	char	   *separator;
+	char	   *after_prefix;
+
+	separator = strchr(collcollate, ':');
+
+	/*
+	 * If it's a traditional value without a prefix, use the default ICU
+	 * library.  That's the one we were linked against, or another one if
+	 * default_icu_library_version has been set.
+	 */
+	if (separator == NULL)
+	{
+		*rest = collcollate;
+
+		if (default_icu_library_version > 0)
+			major_version = default_icu_library_version;
+		else
+			major_version = U_ICU_VERSION_MAJOR_NUM;
+		return get_icu_library(major_version);
+	}
+
+	/* If it has a prefix, interpret it as an ICU major version. */
+	major_version = strtol(collcollate, &after_prefix, 10);
+	if (after_prefix != separator)
+		elog(ERROR,
+			 "could not parse ICU major library version: \"%s\"",
+			 collcollate);
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+		elog(ERROR,
+			 "ICU major library verision out of supported range: \"%s\"",
+			 collcollate);
+
+	/* The part after the separate will be passed to the library. */
+	*rest = separator + 1;
+
+	return get_icu_library(major_version);
+}
+
+#endif
+
 void
 make_icu_collator(const char *iculocstr,
 				  struct pg_locale_struct *resultp)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(iculocstr, &iculocstr);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(iculocstr, &status);
+	collator = lib->open(iculocstr, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						iculocstr, u_errorName(status))));
+						iculocstr, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	/*
+	 * XXX can we just drop this cruft and make 54 the minimum supported
+	 * version?
+	 */
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, iculocstr);
 
 	/* We will leak this string if the caller errors later :-( */
 	resultp->info.icu.locale = MemoryContextStrdup(TopMemoryContext, iculocstr);
 	resultp->info.icu.ucol = collator;
+	resultp->info.icu.lib = lib;
 #else							/* not USE_ICU */
 	/* could get here if a collation was created by a build with ICU */
 	ereport(ERROR,
@@ -1651,21 +1996,23 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 #ifdef USE_ICU
 	if (collprovider == COLLPROVIDER_ICU)
 	{
+		pg_icu_library *lib;
 		UCollator  *collator;
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
 
+		lib = get_icu_library_for_collation(collcollate, &collcollate);
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collcollate, &status);
+		collator = lib->open(collcollate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collcollate, u_errorName(status))));
-		ucol_getVersion(collator, versioninfo);
-		ucol_close(collator);
+							collcollate, lib->errorName(status))));
+		lib->getVersion(collator, versioninfo);
+		lib->close(collator);
 
-		u_versionToString(versioninfo, buf);
+		lib->versionToString(versioninfo, buf);
 		collversion = pstrdup(buf);
 	}
 	else
@@ -1733,6 +2080,8 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 
 
 #ifdef USE_ICU
+
+
 /*
  * Converter object for converting between ICU's UChar strings and C strings
  * in database encoding.  Since the database encoding doesn't change, we only
@@ -1954,19 +2303,22 @@ void
 check_icu_locale(const char *icu_locale)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(icu_locale, &icu_locale);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(icu_locale, &status);
+	collator = lib->open(icu_locale, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						icu_locale, u_errorName(status))));
+						icu_locale, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	/* XXX can we just drop this cruft? */
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, icu_locale);
-	ucol_close(collator);
+	lib->close(collator);
 #else
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index 68e2e6f7a7..e0c86870e0 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -1026,11 +1026,11 @@ hashbpchar(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, keydata, keylen);
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -1087,11 +1087,11 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c5e7ee7ca2..cf891a5654 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1667,13 +1667,14 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					UErrorCode	status;
 
 					status = U_ZERO_ERROR;
-					result = ucol_strcollUTF8(mylocale->info.icu.ucol,
-											  arg1, len1,
-											  arg2, len2,
-											  &status);
+					result = PG_ICU_LIB(mylocale)->strcollUTF8(PG_ICU_COL(mylocale),
+															   arg1, len1,
+															   arg2, len2,
+															   &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
-								(errmsg("collation failed: %s", u_errorName(status))));
+								(errmsg("collation failed: %s",
+										PG_ICU_LIB(mylocale)->errorName(status))));
 				}
 				else
 #endif
@@ -1686,9 +1687,9 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					ulen1 = icu_to_uchar(&uchar1, arg1, len1);
 					ulen2 = icu_to_uchar(&uchar2, arg2, len2);
 
-					result = ucol_strcoll(mylocale->info.icu.ucol,
-										  uchar1, ulen1,
-										  uchar2, ulen2);
+					result = PG_ICU_LIB(mylocale)->strcoll(PG_ICU_COL(mylocale),
+														   uchar1, ulen1,
+														   uchar2, ulen2);
 
 					pfree(uchar1);
 					pfree(uchar2);
@@ -2388,13 +2389,14 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				UErrorCode	status;
 
 				status = U_ZERO_ERROR;
-				result = ucol_strcollUTF8(sss->locale->info.icu.ucol,
-										  a1p, len1,
-										  a2p, len2,
-										  &status);
+				result = PG_ICU_LIB(sss->locale)->strcollUTF8(PG_ICU_COL(sss->locale),
+															  a1p, len1,
+															  a2p, len2,
+															  &status);
 				if (U_FAILURE(status))
 					ereport(ERROR,
-							(errmsg("collation failed: %s", u_errorName(status))));
+							(errmsg("collation failed: %s",
+									PG_ICU_LIB(sss->locale)->errorName(status))));
 			}
 			else
 #endif
@@ -2407,9 +2409,9 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				ulen1 = icu_to_uchar(&uchar1, a1p, len1);
 				ulen2 = icu_to_uchar(&uchar2, a2p, len2);
 
-				result = ucol_strcoll(sss->locale->info.icu.ucol,
-									  uchar1, ulen1,
-									  uchar2, ulen2);
+				result = PG_ICU_LIB(sss->locale)->strcoll(PG_ICU_COL(sss->locale),
+														  uchar1, ulen1,
+														  uchar2, ulen2);
 
 				pfree(uchar1);
 				pfree(uchar2);
@@ -2569,24 +2571,24 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 					uint32_t	state[2];
 					UErrorCode	status;
 
-					uiter_setUTF8(&iter, sss->buf1, len);
+					PG_ICU_LIB(sss->locale)->setUTF8(&iter, sss->buf1, len);
 					state[0] = state[1] = 0;	/* won't need that again */
 					status = U_ZERO_ERROR;
-					bsize = ucol_nextSortKeyPart(sss->locale->info.icu.ucol,
-												 &iter,
-												 state,
-												 (uint8_t *) sss->buf2,
-												 Min(sizeof(Datum), sss->buflen2),
-												 &status);
+					bsize = PG_ICU_LIB(sss->locale)->nextSortKeyPart(PG_ICU_COL(sss->locale),
+																	 &iter,
+																	 state,
+																	 (uint8_t *) sss->buf2,
+																	 Min(sizeof(Datum), sss->buflen2),
+																	 &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
 								(errmsg("sort key generation failed: %s",
-										u_errorName(status))));
+										PG_ICU_LIB(sss->locale)->errorName(status))));
 				}
 				else
-					bsize = ucol_getSortKey(sss->locale->info.icu.ucol,
-											uchar, ulen,
-											(uint8_t *) sss->buf2, sss->buflen2);
+					bsize = PG_ICU_LIB(sss->locale)->getSortKey(PG_ICU_COL(sss->locale),
+																uchar, ulen,
+																(uint8_t *) sss->buf2, sss->buflen2);
 			}
 			else
 #endif
diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c
index 05ab087934..9489268b39 100644
--- a/src/backend/utils/misc/guc_tables.c
+++ b/src/backend/utils/misc/guc_tables.c
@@ -2939,6 +2939,20 @@ struct config_int ConfigureNamesInt[] =
 		check_max_worker_processes, NULL, NULL
 	},
 
+	{
+		{"default_icu_library_version",
+			PGC_SIGHUP,
+			COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Default major version of ICU library to use for collations if not specified."),
+			NULL
+		},
+		&default_icu_library_version,
+		0,
+		0,
+		1000,
+		NULL, NULL, NULL
+	},
+
 	{
 		{"max_logical_replication_workers",
 			PGC_POSTMASTER,
@@ -3922,6 +3936,20 @@ struct config_string ConfigureNamesString[] =
 		NULL, NULL, NULL
 	},
 
+	{
+		{"icu_library_path", PGC_SUSET, COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Sets the path for dynamically loadable ICU libraries."),
+			gettext_noop("If versions of ICU other than the one that "
+						 "PostgreSQL is linked against are needed, they will "
+						 "be opened from this directory.  If empty, the "
+						 "system linker search path will be used."),
+			GUC_SUPERUSER_ONLY
+		},
+		&icu_library_path,
+		"",
+		NULL, NULL, NULL
+	},
+
 	{
 		{"krb_server_keyfile", PGC_SIGHUP, CONN_AUTH_AUTH,
 			gettext_noop("Sets the location of the Kerberos server key file."),
diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample
index 868d21c351..2713c92124 100644
--- a/src/backend/utils/misc/postgresql.conf.sample
+++ b/src/backend/utils/misc/postgresql.conf.sample
@@ -727,6 +727,11 @@
 #lc_numeric = 'C'			# locale for number formatting
 #lc_time = 'C'				# locale for time formatting
 
+#default_icu_library_version = 0	# default major version of ICU library
+					# (0 for the linked version)
+#icu_library_path = ''			# path for dynamically loaded ICU
+					# libraries
+
 # default configuration for text search
 #default_text_search_config = 'pg_catalog.simple'
 
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index a875942123..d26e5738f9 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -17,6 +17,7 @@
 #endif
 #ifdef USE_ICU
 #include <unicode/ucol.h>
+#include <unicode/ubrk.h>
 #endif
 
 #ifdef USE_ICU
@@ -40,6 +41,8 @@ extern PGDLLIMPORT char *locale_messages;
 extern PGDLLIMPORT char *locale_monetary;
 extern PGDLLIMPORT char *locale_numeric;
 extern PGDLLIMPORT char *locale_time;
+extern PGDLLIMPORT char *icu_library_path;
+extern PGDLLIMPORT int default_icu_library_version;
 
 /* lc_time localization cache */
 extern PGDLLIMPORT char *localized_abbrev_days[];
@@ -63,6 +66,71 @@ extern struct lconv *PGLC_localeconv(void);
 
 extern void cache_locale_time(void);
 
+#ifdef USE_ICU
+
+/*
+ * An ICU library version that we're either linked against or have loaded at
+ * runtime.
+ */
+typedef struct pg_icu_library
+{
+	int			major_version;
+	void	   *libicui18n_handle;
+	void	   *libicuuc_handle;
+	UCollator  *(*open) (const char *loc, UErrorCode *status);
+	void		(*close) (UCollator *coll);
+	void		(*getVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*versionToString) (const UVersionInfo versionArray,
+									char *versionString);
+				UCollationResult(*strcoll) (const UCollator *coll,
+											const UChar *source,
+											int32_t sourceLength,
+											const UChar *target,
+											int32_t targetLength);
+				UCollationResult(*strcollUTF8) (const UCollator *coll,
+												const char *source,
+												int32_t sourceLength,
+												const char *target,
+												int32_t targetLength,
+												UErrorCode *status);
+	int32_t		(*getSortKey) (const UCollator *coll,
+							   const UChar *source,
+							   int32_t sourceLength,
+							   uint8_t *result,
+							   int32_t resultLength);
+	int32_t		(*nextSortKeyPart) (const UCollator *coll,
+									UCharIterator *iter,
+									uint32_t state[2],
+									uint8_t *dest,
+									int32_t count,
+									UErrorCode *status);
+	void		(*setUTF8) (UCharIterator *iter,
+							const char *s,
+							int32_t length);
+	const char *(*errorName) (UErrorCode code);
+	int32_t		(*strToUpper) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToLower) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToTitle) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UBreakIterator *titleIter,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	struct pg_icu_library *next;
+} pg_icu_library;
+
+#endif
 
 /*
  * We define our own wrapper around locale_t so we can keep the same
@@ -84,12 +152,18 @@ struct pg_locale_struct
 		{
 			const char *locale;
 			UCollator  *ucol;
+			pg_icu_library *lib;
 		}			icu;
 #endif
 		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
 	}			info;
 };
 
+#ifdef USE_ICU
+#define PG_ICU_LIB(x) ((x)->info.icu.lib)
+#define PG_ICU_COL(x) ((x)->info.icu.ucol)
+#endif
+
 typedef struct pg_locale_struct *pg_locale_t;
 
 extern PGDLLIMPORT struct pg_locale_struct default_locale;
diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list
index d9b839c979..0ccbbb711e 100644
--- a/src/tools/pgindent/typedefs.list
+++ b/src/tools/pgindent/typedefs.list
@@ -1101,6 +1101,7 @@ HeapTupleTableSlot
 HistControl
 HotStandbyState
 I32
+ICU_Convert_BI_Func
 ICU_Convert_Func
 ID
 INFIX
@@ -2854,6 +2855,7 @@ TypeName
 U
 U32
 U8
+UBreakIterator
 UChar
 UCharIterator
 UColAttribute
@@ -3482,6 +3484,7 @@ pg_funcptr_t
 pg_gssinfo
 pg_hmac_ctx
 pg_hmac_errno
+pg_icu_library
 pg_int64
 pg_local_to_utf_combined
 pg_locale_t
-- 
2.30.2



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-01 10:33  Peter Eisentraut <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 1 reply; 57+ messages in thread

From: Peter Eisentraut @ 2022-11-01 10:33 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 22.10.22 03:22, Thomas Munro wrote:
> Suppose your pgdata encounters a PostgreSQL linked against a later ICU
> library, most likely after an OS upgrade or migratoin, a pg_upgrade,
> or via streaming replication.  You might get a new error "can't find
> ICU collation 'en' with version '153.14'; HINT: install missing ICU
> library version", and somehow you'll have to work out which one might
> contain 'en' v153.14 and install it with apt-get etc.  Then it'll
> magically work: your postgres linked against (say) 71 will happily
> work with the dlopen'd 67.  This is enough if you want to stay on 67
> until the heat death of the universe.  So far so good.

What I'm wondering is where those ICU installations are going to come 
from.  In order for this project to be viable, we would need to convince 
some combination of ICU maintainers, OS packagers, and PGDG packagers to 
provide and maintain five year's worth of ICU packages (yearly releases 
AFAICT).  Is that something we are willing to get into?

(Even to test this I need to figure out where to get another ICU 
installation from.  I'll try how easy manual installations are.)






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-01 12:42  Thomas Munro <[email protected]>
  parent: Peter Eisentraut <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-01 12:42 UTC (permalink / raw)
  To: Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 1, 2022 at 11:33 PM Peter Eisentraut
<[email protected]> wrote:
> What I'm wondering is where those ICU installations are going to come
> from.  In order for this project to be viable, we would need to convince
> some combination of ICU maintainers, OS packagers, and PGDG packagers to
> provide and maintain five year's worth of ICU packages (yearly releases
> AFAICT).  Is that something we are willing to get into?

I hacked on this on a Debian machine that has a couple of these
installed and they work fine, but now I realise that might have to do
with the major upgrade history of the machine.  So yeah... probably.
:-/  Not being involved in packaging I have no idea how plausible such
a backports (erm, forwardports?) repo would be, and I have even less
idea for other distros.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-01 23:57  Thomas Munro <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-01 23:57 UTC (permalink / raw)
  To: Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 2, 2022 at 1:42 AM Thomas Munro <[email protected]> wrote:
> On Tue, Nov 1, 2022 at 11:33 PM Peter Eisentraut
> <[email protected]> wrote:
> > What I'm wondering is where those ICU installations are going to come
> > from.  In order for this project to be viable, we would need to convince
> > some combination of ICU maintainers, OS packagers, and PGDG packagers to
> > provide and maintain five year's worth of ICU packages (yearly releases
> > AFAICT).  Is that something we are willing to get into?
>
> I hacked on this on a Debian machine that has a couple of these
> installed and they work fine, but now I realise that might have to do
> with the major upgrade history of the machine.  So yeah... probably.
> :-/  Not being involved in packaging I have no idea how plausible such
> a backports (erm, forwardports?) repo would be, and I have even less
> idea for other distros.

After sleeping on it, I don't really agree that the project is not
viable even if it requires hoop-jumping to set up right now.  It's a
chicken-and-egg problem, and the first step is to make it possible to
do it at all, thereby creating the demand for convenient packages.  I
think we have several topics here:

1.  Technical problems relating to dlopen'ing.  Does it work?  Is the
default dlopen() secure enough?  Is it building sensible library
names, even on the freaky-library OSes (Windows, macOS, AIX)?  Is it
enough to have that GUC for non-default path, should it be a search
path, should it share the existing dynamic_library_path?  Are the
indirect function calls fast enough?  Is the way it handles API
stability sound?  Can we drop some unfinished complexity by dropping
pre-53 ICU?  Does it use too much memory?
2.  User experience problems relating to upgrade paths and user
interface.  Is it enough to start with the basic DB2-style approach
that I've prototyped here?  How should we refer to library versions?
Is your search-for-the-collversion idea better?  My gut feeling is
that the early version should be about giving people options, and not
trying to be too clever/automatic with questionable semantics, and
later improvements could follow, for example if we have another go at
the per-object version tracking.
3.  Library availability.  This is a problem for downstream
communities to solve.  For example, the people who build Windows
installers might want to start bundling the ICU versions from their
earlier releases, the people involved with each Linux/BSD distro would
hopefully figure out a good way to publish the packages from older OS
releases in one repo, and the people running managed systems probably
do their own packaging anyway, they'll figure it out.  I realise that
you are involved in packaging and I am not, so we probably have
different perspectives: I get to say "and here, magic happens!" :-)

FWIW at least 57, 63 and 67 (corresponding to deb9, 10, 11) from
http://ftp.debian.org/debian/pool/main/i/icu/ can be installed with
dpkg -i on my Debian 11 machine.  52 (deb8) too, probably, but it has
dependencies I didn't look into.  71 and 72 are newer than the -dev
version (what we link against), so I didn't try installing but the
patch as posted wouldn't let me open them: the idea here is to allow
only older stuff to be dlopen'd, so if a breaking API change comes
down the pipe we'll be able to deal with it.  Not being a packaging
guy, I don't how how stupid it would be to build a package repo that
literally just exposes these via an index and that's all, or whether
it's better to rebuild the ICU versions from source against modern
C/C++ runtimes etc.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-07 12:21  Peter Eisentraut <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Peter Eisentraut @ 2022-11-07 12:21 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 02.11.22 00:57, Thomas Munro wrote:
> 3.  Library availability.  This is a problem for downstream
> communities to solve.  For example, the people who build Windows
> installers might want to start bundling the ICU versions from their
> earlier releases, the people involved with each Linux/BSD distro would
> hopefully figure out a good way to publish the packages from older OS
> releases in one repo, and the people running managed systems probably
> do their own packaging anyway, they'll figure it out.  I realise that
> you are involved in packaging and I am not, so we probably have
> different perspectives: I get to say "and here, magic happens!" :-)

I made a Homebrew repository for ICU versions 50 through 72: 
https://github.com/petere/homebrew-icu

All of these packages build and pass their self-tests on my machine.  So 
from that experience, I think maintaining a repository of ICU versions, 
and being able to install more than one for testing this feature, is 
feasible.

Now I have started building PostgreSQL against these, to get some 
baseline of what is supported and actually works.  The results are a bit 
mixed so far, more to come later.

The installation instructions currently say that the minimum required 
version of ICU is 4.2.  That was the one that shipped with RHEL 6.  I 
think we have de-supported RHEL 6 and could increase that.  The version 
in RHEL 7 is 50.

(My repository happens to start at 50 because the new versioning system 
started at 49, but 49 doesn't appear to be tagged at the icu github site.)

Note: Recent versions of libxml2 link against icu.  This isn't a 
problem, thanks to the symbol versioning, but if you get libxml2 via 
pkg-config, you might get LDFLAGS from not the icu version you wanted.






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-09 02:37  Thomas Munro <[email protected]>
  parent: Peter Eisentraut <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-09 02:37 UTC (permalink / raw)
  To: Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 8, 2022 at 1:22 AM Peter Eisentraut
<[email protected]> wrote:
> I made a Homebrew repository for ICU versions 50 through 72:
> https://github.com/petere/homebrew-icu

Nice!

> All of these packages build and pass their self-tests on my machine.  So
> from that experience, I think maintaining a repository of ICU versions,
> and being able to install more than one for testing this feature, is
> feasible.

I wonder what the situation with CVEs is in older releases.  I heard a
rumour that upstream might only patch current + previous, leaving it
up to distros to back-patch to whatever they need to support, but I
haven't tried to track down cold hard evidence of this or think about
what it means for this project...





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-11 14:57  Peter Eisentraut <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 0 replies; 57+ messages in thread

From: Peter Eisentraut @ 2022-11-11 14:57 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 22.10.22 03:22, Thomas Munro wrote:
> I'd love to hear others' thoughts on how we can turn this into a
> workable solution.  Hopefully while staying simple...

I played with this patch a bit.  It looks like a reasonable approach.

Attached is a small patch to get the dynamic libicu* lookup working with 
the library naming on macOS.

Instead of packing the ICU version into the locale field ('63:en'), I 
would make it a separate field in pg_collation and a separate argument 
in CREATE COLLATION.

At this point, perhaps it would be good to start building some tests to 
demonstrate various upgrade scenarios and to ensure portability.


From e236f5257bf0bf3e7b83b9d9b095d1d0e3fdc971 Mon Sep 17 00:00:00 2001
From: Peter Eisentraut <[email protected]>
Date: Fri, 11 Nov 2022 15:44:44 +0100
Subject: [PATCH] fixup! WIP: Multi-version ICU.

---
 src/backend/utils/adt/pg_locale.c | 8 ++++++++
 1 file changed, 8 insertions(+)

diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 666a79b907a4..3ffb9706ff99 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -1600,6 +1600,10 @@ get_icu_library(int major_version)
 				 "%s%sicui18n%d." DLSUFFIX,
 				 icu_library_path,
 				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicui18n.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
 #else
 				 "%s%slibicui18n" DLSUFFIX ".%d",
 				 icu_library_path,
@@ -1615,6 +1619,10 @@ get_icu_library(int major_version)
 				 "%s%sicuuc%d." DLSUFFIX,
 				 icu_library_path,
 				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicuuc.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
 #else
 				 "%s%slibicuuc" DLSUFFIX ".%d",
 				 icu_library_path,
-- 
2.38.1



Attachments:

  [text/plain] 0001-fixup-WIP-Multi-version-ICU.patch (1.2K, ../../[email protected]/2-0001-fixup-WIP-Multi-version-ICU.patch)
  download | inline diff:
From e236f5257bf0bf3e7b83b9d9b095d1d0e3fdc971 Mon Sep 17 00:00:00 2001
From: Peter Eisentraut <[email protected]>
Date: Fri, 11 Nov 2022 15:44:44 +0100
Subject: [PATCH] fixup! WIP: Multi-version ICU.

---
 src/backend/utils/adt/pg_locale.c | 8 ++++++++
 1 file changed, 8 insertions(+)

diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 666a79b907a4..3ffb9706ff99 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -1600,6 +1600,10 @@ get_icu_library(int major_version)
 				 "%s%sicui18n%d." DLSUFFIX,
 				 icu_library_path,
 				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicui18n.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
 #else
 				 "%s%slibicui18n" DLSUFFIX ".%d",
 				 icu_library_path,
@@ -1615,6 +1619,10 @@ get_icu_library(int major_version)
 				 "%s%sicuuc%d." DLSUFFIX,
 				 icu_library_path,
 				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicuuc.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
 #else
 				 "%s%slibicuuc" DLSUFFIX ".%d",
 				 icu_library_path,
-- 
2.38.1



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-15 00:55  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-15 00:55 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

I looked at v6.

  * We'll need some clearer instructions on how to build/install extra
ICU versions that might not be provided by the distribution packaging.
For instance, I got a cryptic error until I used --enable-rpath, which
might not be obvious to all users.
  * Can we have a better error when the library was built with --
disable-renaming? We can just search for the plain (no suffix) symbol.
  * We should use dlerror() instead of %m to report dlopen() errors.
  * It seems like the collation version is just there to issue WARNINGs
when a user is using the non-versioned locale syntax and the library
changes underneath them (or if there is collation version change within
a single ICU major version)?
  * How are you testing this?
  * In my tests (sort, hacked so abbreviate is always false), I see a
~3% regression for ICU+UTF8. That's fine with me. I assume it's due to
the indirect function call, but that's not obvious to me from the
profile. If it's a major problem we could have a special case of
varstrfastcmp_locale() that works on the compile-time ICU version.

I realize your patch is experimental, but when there is a better
consensus on the approach, we should consider adding declarative syntax
such as:

   CREATE COLLATION (or LOCALE?) PROVIDER icu67
     TYPE icu VERSION '67' AS '/path/to/icui18n.so.67';

It will offer more opportunities to catch errors early and offer better
error messages. It would also enable it to function if the library is
built with --disable-renaming (though we'd have to trust the user).

On Sat, 2022-10-22 at 14:22 +1300, Thomas Munro wrote:
> Problem 1:  Suppose you're ready to start using (say) v72.  I guess
> you'd use the REFRESH command, which would open the main linked ICU's
> collversion and stamp that into the catalogue, at which point new
> sessions would start using that, and then you'd have to rebuild all
> your indexes (with no help from PG to tell you how to find everything
> that needs to be rebuilt, as belaboured in previous reverted work).
> Aside from the possibility of getting the rebuilding job wrong (as
> belaboured elsewhere), it's not great, because there is still a
> transitional period where you can be using the wrong version for your
> data.  So this requires some careful planning and understanding from
> the administrator.

How is this related to the search-by-collversion design? It seems like
it's hard no matter what.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-18 18:38  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 2 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-18 18:38 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

Replying to Peter and Jeff in one email.

On Sat, Nov 12, 2022 at 3:57 AM Peter Eisentraut
<[email protected]> wrote:
> On 22.10.22 03:22, Thomas Munro wrote:
> > I'd love to hear others' thoughts on how we can turn this into a
> > workable solution.  Hopefully while staying simple...
>
> I played with this patch a bit.  It looks like a reasonable approach.

Great news.

> Attached is a small patch to get the dynamic libicu* lookup working with
> the library naming on macOS.

Thanks, squashed.

> Instead of packing the ICU version into the locale field ('63:en'), I
> would make it a separate field in pg_collation and a separate argument
> in CREATE COLLATION.

I haven't tried this yet, as I focused on coming up with a way of testing in
this iteration.  I can try this next.  I'm imagining that we'd have
pg_collation.collicuversion and pg_database.daticuversion, and they'd default
to 0 for "use the GUC", and perhaps you'd even be able to ALTER them.  Perhaps
we wouldn't even need the GUC then...  0 could mean "the linked version", and
if you don't like it, you ALTER it.  Thinking about this.

> At this point, perhaps it would be good to start building some tests to
> demonstrate various upgrade scenarios and to ensure portability.

OK, here's what I came up with.  You enable it in PG_TEST_EXTRA, and
tell it about an alternative ICU version you have in the standard library
search path that is not the same as the main/linked one:

$ meson configure -DPG_TEST_EXTRA="icu=63"
$ meson test icu/020_multiversion

Another change from your feedback:  you mentioned that RHEL7 shipped with ICU
50, so I removed my suggestion of dropping some extra code we carry for
versions before 54 and set the minimum acceptable version to 50.  It probably
works further back than that, but that's a decent range, I think.

On Tue, Nov 15, 2022 at 1:55 PM Jeff Davis <[email protected]> wrote:
> I looked at v6.

Thanks for jumping in and testing!

>   * We'll need some clearer instructions on how to build/install extra
> ICU versions that might not be provided by the distribution packaging.
> For instance, I got a cryptic error until I used --enable-rpath, which
> might not be obvious to all users.

Suggestions welcome.  No docs at all yet...

>   * Can we have a better error when the library was built with --
> disable-renaming? We can just search for the plain (no suffix) symbol.

I threw out that symbol probing logic, and wrote something simpler that should
now also work with --disable-renaming (though not tested).  Now it does a
cross-check with the library's self-reported major version, just to make
sure there wasn't a badly named library file, which may be more likely
with --disable-renaming.

>   * We should use dlerror() instead of %m to report dlopen() errors.

Fixed.

>   * It seems like the collation version is just there to issue WARNINGs
> when a user is using the non-versioned locale syntax and the library
> changes underneath them (or if there is collation version change within
> a single ICU major version)?

Correct.

I have now updated the warning messages you get when they don't match, to
provide a hint about what to do about it.  I am sure they need some more
word-smithing, though.

>   * How are you testing this?

Ad hoc noodling before now, but see attached.

> I realize your patch is experimental, but when there is a better
> consensus on the approach, we should consider adding declarative syntax
> such as:
>
>    CREATE COLLATION (or LOCALE?) PROVIDER icu67
>      TYPE icu VERSION '67' AS '/path/to/icui18n.so.67';
>
> It will offer more opportunities to catch errors early and offer better
> error messages. It would also enable it to function if the library is
> built with --disable-renaming (though we'd have to trust the user).

Earlier in this and other threads, we wondered if each ICU major version should
be a separate provider, which is what you're showing there, or should be an
independent property of an individual COLLATION, which is what v6 did with
'63:en' and what Peter suggested I make more formal with CREATE COLLATION foo
(..., ICU_VERSION=63).  I actually started out thinking we'd have multiple
providers, but I couldn't really think of any advantage, and I think it makes
some upgrade scenarios more painful.  Can you elaborate on why you'd want
that model?

> On Sat, 2022-10-22 at 14:22 +1300, Thomas Munro wrote:
> > Problem 1:  Suppose you're ready to start using (say) v72.  I guess
> > you'd use the REFRESH command, which would open the main linked ICU's
> > collversion and stamp that into the catalogue, at which point new
> > sessions would start using that, and then you'd have to rebuild all
> > your indexes (with no help from PG to tell you how to find everything
> > that needs to be rebuilt, as belaboured in previous reverted work).
> > Aside from the possibility of getting the rebuilding job wrong (as
> > belaboured elsewhere), it's not great, because there is still a
> > transitional period where you can be using the wrong version for your
> > data.  So this requires some careful planning and understanding from
> > the administrator.
>
> How is this related to the search-by-collversion design? It seems like
> it's hard no matter what.

Yeah.  I just don't like the way it *appears* to be doing something clever, but
it doesn't solve any fundamental problem at all because the collversion
information is under human control and so it's really doing something stupid.
Hence desire to build something that at least admits that it's primitive and
just gives you some controls, in a first version.  We could always reconsider
that in later work though, maybe even an optional policy or something?


Attachments:

  [text/x-patch] v7-0001-WIP-Multi-version-ICU.patch (50.5K, ../../CA+hUKG+OSQtrRAk-bHwMJmuvqp-b-LGuxsF2PoDBVyQcT+VEAQ@mail.gmail.com/2-v7-0001-WIP-Multi-version-ICU.patch)
  download | inline diff:
From 51f0e2eaaf8e941033ad4ba7e412fc900636962d Mon Sep 17 00:00:00 2001
From: Thomas Munro <[email protected]>
Date: Wed, 8 Jun 2022 17:43:53 +1200
Subject: [PATCH v7] WIP: Multi-version ICU.

Add a layer of indirection when accessing ICU, so that multiple major
versions of the library can be used at once.  Versions other than the
one that PostgreSQL was linked against are opened with dlopen(), but we
refuse to open version higher than the one were were compiled against.
The ABI might change in future releases so that wouldn't be safe.

By default, the system linker's default search path is used to find
libraries, but icu_library_path may be used to specify an absolute path
to look in.  ICU libraries are expected to have been built without ICU's
--disable-renaming option.  That is, major versions must use distinct
symbol names.

This arrangement means that at least one major version of ICU is always
available -- the one that PostgreSQL was linked again.  It should be
simple on most software distributions to install extra versions using a
package manager, or to build extra libraries as required, to access
older ICU releases.  For example, on Debian bullseye the packages are
named libicu63, libicu67, libicu71.

In this version of the patch, '63:en' used as a database default locale
or COLLATION object requests ICU library 63, and 'en' requests the
library version seleted by the GUC default_icu_library_version,
defaulting to the version that the executable is linked against.

XXX Many other designs possible, to discuss!

Reviewed-by: Peter Eisentraut <[email protected]>
Reviewed-by: Jeff Davis <[email protected]>
Discussion: https://postgr.es/m/CA%2BhUKGL4VZRpP3CkjYQkv4RQ6pRYkPkSNgKSxFBwciECQ0mEuQ%40mail.gmail.com
---
 src/backend/access/hash/hashfunc.c            |  16 +-
 src/backend/commands/collationcmds.c          |  20 +
 src/backend/utils/adt/formatting.c            |  53 +-
 src/backend/utils/adt/pg_locale.c             | 451 +++++++++++++++++-
 src/backend/utils/adt/varchar.c               |  16 +-
 src/backend/utils/adt/varlena.c               |  56 +--
 src/backend/utils/init/postinit.c             |  48 +-
 src/backend/utils/misc/guc_tables.c           |  28 ++
 src/backend/utils/misc/postgresql.conf.sample |   5 +
 src/include/catalog/pg_proc.dat               |   3 +
 src/include/utils/pg_locale.h                 |  75 +++
 src/test/icu/meson.build                      |   1 +
 src/test/icu/t/020_multiversion.pl            | 203 ++++++++
 src/tools/pgindent/typedefs.list              |   3 +
 14 files changed, 888 insertions(+), 90 deletions(-)
 create mode 100644 src/test/icu/t/020_multiversion.pl

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index b57ed946c4..0a61538efd 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -298,11 +298,11 @@ hashtext(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -355,11 +355,11 @@ hashtextextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index 81e54e0ce6..4fb0c77f38 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -853,6 +853,26 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 					CreateComments(collid, CollationRelationId, 0,
 								   icucomment);
 			}
+
+			/* Also create an object pinned to an ICU major version. */
+			collid = CollationCreate(psprintf("%s-x-icu-%d", langtag, U_ICU_VERSION_MAJOR_NUM),
+									 nspid, GetUserId(),
+									 COLLPROVIDER_ICU, true, -1,
+									 NULL, NULL,
+									 psprintf("%d:%s", U_ICU_VERSION_MAJOR_NUM, iculocstr),
+									 get_collation_actual_version(COLLPROVIDER_ICU, iculocstr),
+									 true, true);
+			if (OidIsValid(collid))
+			{
+				ncreated++;
+
+				CommandCounterIncrement();
+
+				icucomment = get_icu_locale_comment(name);
+				if (icucomment)
+					CreateComments(collid, CollationRelationId, 0,
+								   icucomment);
+			}
 		}
 	}
 #endif							/* USE_ICU */
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index 26f498b5df..0c3c7724d7 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -1599,6 +1599,11 @@ typedef int32_t (*ICU_Convert_Func) (UChar *dest, int32_t destCapacity,
 									 const UChar *src, int32_t srcLength,
 									 const char *locale,
 									 UErrorCode *pErrorCode);
+typedef int32_t (*ICU_Convert_BI_Func) (UChar *dest, int32_t destCapacity,
+										const UChar *src, int32_t srcLength,
+										UBreakIterator *bi,
+										const char *locale,
+										UErrorCode *pErrorCode);
 
 static int32_t
 icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
@@ -1623,18 +1628,41 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("case conversion failed: %s", u_errorName(status))));
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
 	return len_dest;
 }
 
+/*
+ * Like icu_convert_case, but func takes a break iterator (which we don't
+ * make use of).
+ */
 static int32_t
-u_strToTitle_default_BI(UChar *dest, int32_t destCapacity,
-						const UChar *src, int32_t srcLength,
-						const char *locale,
-						UErrorCode *pErrorCode)
+icu_convert_case_bi(ICU_Convert_BI_Func func, pg_locale_t mylocale,
+					UChar **buff_dest, UChar *buff_source, int32_t len_source)
 {
-	return u_strToTitle(dest, destCapacity, src, srcLength,
-						NULL, locale, pErrorCode);
+	UErrorCode	status;
+	int32_t		len_dest;
+
+	len_dest = len_source;		/* try first with same length */
+	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+	status = U_ZERO_ERROR;
+	len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+					mylocale->info.icu.locale, &status);
+	if (status == U_BUFFER_OVERFLOW_ERROR)
+	{
+		/* try again with adjusted length */
+		pfree(*buff_dest);
+		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+		status = U_ZERO_ERROR;
+		len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+						mylocale->info.icu.locale, &status);
+	}
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
+	return len_dest;
 }
 
 #endif							/* USE_ICU */
@@ -1702,7 +1730,8 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToLower, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToLower,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1824,7 +1853,8 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToUpper, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToUpper,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1947,8 +1977,9 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToTitle_default_BI, mylocale,
-										&buff_conv, buff_uchar, len_uchar);
+			len_conv = icu_convert_case_bi(PG_ICU_LIB(mylocale)->strToTitle,
+										   mylocale,
+										   &buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 2b42d9ccd8..3cc51a54a8 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -58,6 +58,7 @@
 #include "catalog/pg_collation.h"
 #include "catalog/pg_control.h"
 #include "mb/pg_wchar.h"
+#include "miscadmin.h"
 #include "utils/builtins.h"
 #include "utils/formatting.h"
 #include "utils/guc_hooks.h"
@@ -69,6 +70,7 @@
 
 #ifdef USE_ICU
 #include <unicode/ucnv.h>
+#include <unicode/ustring.h>
 #endif
 
 #ifdef __GLIBC__
@@ -79,14 +81,35 @@
 #include <shlwapi.h>
 #endif
 
+#include <dlfcn.h>
+
 #define		MAX_L10N_DATA		80
 
+#ifdef USE_ICU
+
+/*
+ * We don't want to call into dlopen'd ICU libraries that are newer than the
+ * one we were compiled and linked against, just in case there is an
+ * incompatible API change.
+ */
+#define PG_MAX_ICU_MAJOR_VERSION U_ICU_VERSION_MAJOR_NUM
+
+/*
+ * The oldest ICU release we're likely to encounter, and that has all the
+ * funcitons required.
+ */
+#define PG_MIN_ICU_MAJOR_VERSION 50
+
+#endif
+
 
 /* GUC settings */
 char	   *locale_messages;
 char	   *locale_monetary;
 char	   *locale_numeric;
 char	   *locale_time;
+char	   *icu_library_path;
+int			default_icu_library_version;
 
 /*
  * lc_time localization cache.
@@ -1398,29 +1421,348 @@ lc_ctype_is_c(Oid collation)
 	return (lookup_collation_cache(collation, true))->ctype_is_c;
 }
 
+#ifdef USE_ICU
+
 struct pg_locale_struct default_locale;
 
+/* Linked list of ICU libraries we have loaded. */
+static pg_icu_library *icu_library_list = NULL;
+
+/*
+ * Free an ICU library.  pg_icu_library objects that are successfully
+ * constructed stick around for the lifetime of the backend, but this is used
+ * to clean up if initialization fails.
+ */
+static void
+free_icu_library(pg_icu_library *lib)
+{
+	if (lib->libicui18n_handle)
+		dlclose(lib->libicui18n_handle);
+	if (lib->libicuuc_handle)
+		dlclose(lib->libicuuc_handle);
+	pfree(lib);
+}
+
+static void *
+get_icu_function(void *handle, const char *function, int version)
+{
+	char		function_with_version[80];
+	void	   *result;
+
+	/*
+	 * Try to look it up using the symbols with major versions, but if that
+	 * doesn't work, also try the unversioned name in case the library was
+	 * configured with --disable-renaming.
+	 */
+	snprintf(function_with_version, sizeof(function_with_version), "%s_%d",
+			 function, version);
+	result = dlsym(handle, function_with_version);
+
+	return result ? result : dlsym(handle, function);
+}
+
+/*
+ * Helper to load a library.
+ */
+static void *
+load_icu_library(pg_icu_library *lib, const char *name)
+{
+	void	   *handle;
+
+	handle = dlopen(name, RTLD_NOW | RTLD_GLOBAL);
+	if (handle == NULL)
+	{
+		char		message[80];
+
+		strlcpy(message, dlerror(), sizeof(message));
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("could not load library \"%s\": %s", name, message)));
+	}
+
+	return handle;
+}
+
+/*
+ * Given an ICU major version number, return the object we need to access it,
+ * or fail while trying to load it.
+ */
+static pg_icu_library *
+get_icu_library(int major_version)
+{
+	UVersionInfo versioninfo;
+	char		versioninfostring[U_MAX_VERSION_STRING_LENGTH];
+	pg_icu_library *lib;
+
+	/* XXX Move range check into guc_table.c? */
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+		elog(ERROR,
+			"ICU version must be between %d and %d",
+			 PG_MIN_ICU_MAJOR_VERSION,
+			 PG_MAX_ICU_MAJOR_VERSION);
+
+	/* Try to find it in our list of existing libraries. */
+	for (lib = icu_library_list; lib; lib = lib->next)
+		if (lib->major_version == major_version)
+			return lib;
+
+	/* Make a new entry. */
+	lib = MemoryContextAllocZero(TopMemoryContext, sizeof(*lib));
+	if (major_version == U_ICU_VERSION_MAJOR_NUM)
+	{
+		/*
+		 * This is the version we were compiled and linked against.  Simply
+		 * assign the function pointers.
+		 *
+		 * These assignments will fail to compile if an incompatible API
+		 * change is made to some future version of ICU, at which point we
+		 * might need to consider special treatment for different major
+		 * version ranges, with intermediate trampoline functions.
+		 */
+		lib->major_version = major_version;
+		lib->getLibraryVersion = u_getVersion;
+		lib->open = ucol_open;
+		lib->close = ucol_close;
+		lib->getVersion = ucol_getVersion;
+		lib->versionToString = u_versionToString;
+		lib->strcoll = ucol_strcoll;
+		lib->strcollUTF8 = ucol_strcollUTF8;
+		lib->getSortKey = ucol_getSortKey;
+		lib->nextSortKeyPart = ucol_nextSortKeyPart;
+		lib->setUTF8 = uiter_setUTF8;
+		lib->errorName = u_errorName;
+		lib->strToUpper = u_strToUpper;
+		lib->strToLower = u_strToLower;
+		lib->strToTitle = u_strToTitle;
+
+		/*
+		 * Also assert the size of a couple of types used as output buffers,
+		 * as a canary to tell us to add extra padding in the (unlikely) event
+		 * that a later release makes these values smaller.
+		 */
+		StaticAssertStmt(U_MAX_VERSION_STRING_LENGTH == 20,
+						 "u_versionToString output buffer size changed incompatibly");
+		StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
+						 "ucol_getVersion output buffer size changed incompatibly");
+	}
+	else
+	{
+		/* This is an older version, so we'll need to use dlopen(). */
+		char		libicui18n_name[MAXPGPATH];
+		char		libicuuc_name[MAXPGPATH];
+
+		/*
+		 * We don't like to open versions newer than what we're linked
+		 * against, to reduce the risk of an API change biting us.
+		 */
+		if (major_version > U_ICU_VERSION_MAJOR_NUM)
+			elog(ERROR, "ICU major version %d higher than linked version %d, refusing to open",
+				 major_version, U_ICU_VERSION_MAJOR_NUM);
+
+		lib->major_version = major_version;
+
+		/*
+		 * See
+		 * https://unicode-org.github.io/icu/userguide/icu4c/packaging.html#icu-versions
+		 * for conventions on library naming on POSIX and Windows systems.
+		 */
+
+		/* Load the collation library. */
+		snprintf(libicui18n_name,
+				 sizeof(libicui18n_name),
+#ifdef WIN32
+				 "%s%sicui18n%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicui18n.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#else
+				 "%s%slibicui18n" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicui18n_handle = load_icu_library(lib, libicui18n_name);
+
+		/* Load the ctype library. */
+		snprintf(libicuuc_name,
+				 sizeof(libicuuc_name),
+#ifdef WIN32
+				 "%s%sicuuc%d." DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "\\" : "",
+#elif defined(__darwin__)
+				 "%s%slibicuuc.%d" DLSUFFIX,
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#else
+				 "%s%slibicuuc" DLSUFFIX ".%d",
+				 icu_library_path,
+				 icu_library_path[0] ? "/" : "",
+#endif
+				 major_version);
+		lib->libicuuc_handle = load_icu_library(lib, libicuuc_name);
+
+		/* Look up all the functions we need. */
+		lib->getLibraryVersion = get_icu_function(lib->libicui18n_handle,
+												  "u_getVersion",
+												  major_version);
+		lib->open = get_icu_function(lib->libicui18n_handle,
+									 "ucol_open",
+									 major_version);
+		lib->close = get_icu_function(lib->libicui18n_handle,
+									  "ucol_close",
+									  major_version);
+		lib->getVersion = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getVersion",
+										   major_version);
+		lib->versionToString = get_icu_function(lib->libicui18n_handle,
+												"u_versionToString",
+												major_version);
+		lib->strcoll = get_icu_function(lib->libicui18n_handle,
+										"ucol_strcoll",
+										major_version);
+		lib->strcollUTF8 = get_icu_function(lib->libicui18n_handle,
+											"ucol_strcollUTF8",
+											major_version);
+		lib->getSortKey = get_icu_function(lib->libicui18n_handle,
+										   "ucol_getSortKey",
+										   major_version);
+		lib->nextSortKeyPart = get_icu_function(lib->libicui18n_handle,
+												"ucol_nextSortKeyPart",
+												major_version);
+		lib->setUTF8 = get_icu_function(lib->libicui18n_handle,
+										"uiter_setUTF8",
+										major_version);
+		lib->errorName = get_icu_function(lib->libicui18n_handle,
+										  "u_errorName",
+										  major_version);
+		lib->strToUpper = get_icu_function(lib->libicuuc_handle,
+										   "u_strToUpper",
+										   major_version);
+		lib->strToLower = get_icu_function(lib->libicuuc_handle,
+										   "u_strToLower",
+										   major_version);
+		lib->strToTitle = get_icu_function(lib->libicuuc_handle,
+										   "u_strToTitle",
+										   major_version);
+		if (!lib->getLibraryVersion ||
+			!lib->open ||
+			!lib->close ||
+			!lib->getVersion ||
+			!lib->versionToString ||
+			!lib->strcoll ||
+			!lib->strcollUTF8 ||
+			!lib->getSortKey ||
+			!lib->nextSortKeyPart ||
+			!lib->setUTF8 ||
+			!lib->errorName ||
+			!lib->strToUpper ||
+			!lib->strToLower ||
+			!lib->strToTitle)
+		{
+			free_icu_library(lib);
+			ereport(ERROR,
+					(errmsg("could not find expected symbols in libraries \"%s\" and \"%s\"",
+							libicui18n_name, libicuuc_name)));
+		}
+	}
+
+	/*
+	 * Check that the library's own u_getVersion() function reports the version
+	 * that we expected.  By using atoi() we take only the major part.
+	 */
+	lib->getLibraryVersion(versioninfo);
+	lib->versionToString(versioninfo, versioninfostring);
+	if (atoi(versioninfostring) != major_version)
+	{
+		free_icu_library(lib);
+		ereport(ERROR,
+				(errmsg("opened ICU library with major version %d but it reported its own version as %s",
+						major_version, versioninfostring)));
+	}
+
+	lib->next = icu_library_list;
+	icu_library_list = lib;
+
+	return lib;
+}
+
+/*
+ * Look up the library to use for a given collcollate string.
+ */
+static pg_icu_library *
+get_icu_library_for_collation(const char *collcollate, const char **rest)
+{
+	int			major_version;
+	char	   *separator;
+	char	   *after_prefix;
+
+	separator = strchr(collcollate, ':');
+
+	/*
+	 * If it's a traditional value without a prefix, use the default ICU
+	 * library.  That's the one we were linked against, or another one if
+	 * default_icu_library_version has been set.
+	 */
+	if (separator == NULL)
+	{
+		*rest = collcollate;
+
+		if (default_icu_library_version > 0)
+			major_version = default_icu_library_version;
+		else
+			major_version = U_ICU_VERSION_MAJOR_NUM;
+		return get_icu_library(major_version);
+	}
+
+	/* If it has a prefix, interpret it as an ICU major version. */
+	major_version = strtol(collcollate, &after_prefix, 10);
+	if (after_prefix != separator)
+		elog(ERROR,
+			 "could not parse ICU major library version: \"%s\"",
+			 collcollate);
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+		elog(ERROR,
+			 "ICU major library verision out of supported range: \"%s\"",
+			 collcollate);
+
+	/* The part after the separate will be passed to the library. */
+	*rest = separator + 1;
+
+	return get_icu_library(major_version);
+}
+
+#endif
+
 void
 make_icu_collator(const char *iculocstr,
 				  struct pg_locale_struct *resultp)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(iculocstr, &iculocstr);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(iculocstr, &status);
+	collator = lib->open(iculocstr, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						iculocstr, u_errorName(status))));
+						iculocstr, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, iculocstr);
 
 	/* We will leak this string if the caller errors later :-( */
 	resultp->info.icu.locale = MemoryContextStrdup(TopMemoryContext, iculocstr);
 	resultp->info.icu.ucol = collator;
+	resultp->info.icu.lib = lib;
 #else							/* not USE_ICU */
 	/* could get here if a collation was created by a build with ICU */
 	ereport(ERROR,
@@ -1593,14 +1935,15 @@ pg_newlocale_from_collation(Oid collid)
 		{
 			char	   *actual_versionstr;
 			char	   *collversionstr;
+			char	   *locale;
 
 			collversionstr = TextDatumGetCString(datum);
 
 			datum = SysCacheGetAttr(COLLOID, tp, collform->collprovider == COLLPROVIDER_ICU ? Anum_pg_collation_colliculocale : Anum_pg_collation_collcollate, &isnull);
 			Assert(!isnull);
+			locale = TextDatumGetCString(datum);
 
-			actual_versionstr = get_collation_actual_version(collform->collprovider,
-															 TextDatumGetCString(datum));
+			actual_versionstr = get_collation_actual_version(collform->collprovider, locale);
 			if (!actual_versionstr)
 			{
 				/*
@@ -1614,17 +1957,44 @@ pg_newlocale_from_collation(Oid collid)
 			}
 
 			if (strcmp(actual_versionstr, collversionstr) != 0)
-				ereport(WARNING,
-						(errmsg("collation \"%s\" has version mismatch",
-								NameStr(collform->collname)),
-						 errdetail("The collation in the database was created using version %s, "
-								   "but the operating system provides version %s.",
-								   collversionstr, actual_versionstr),
-						 errhint("Rebuild all objects affected by this collation and run "
-								 "ALTER COLLATION %s REFRESH VERSION, "
-								 "or build PostgreSQL with the right library version.",
-								 quote_qualified_identifier(get_namespace_name(collform->collnamespace),
-															NameStr(collform->collname)))));
+			{
+				if (collform->collprovider == COLLPROVIDER_ICU)
+				{
+					ereport(WARNING,
+							(errmsg("collation \"%s\" has version mismatch",
+									NameStr(collform->collname)),
+							 errdetail("The collation in the database was created using version %s, "
+									   "but the ICU library provides version %s.",
+									   collversionstr, actual_versionstr),
+							 strchr(locale, ':') != NULL ?
+							 errhint("Rebuild all objects affected by this collation and run "
+									 "ALTER COLLATION %s REFRESH VERSION, "
+									 "or build PostgreSQL with the right library version.",
+									 quote_qualified_identifier(get_namespace_name(collform->collnamespace),
+																NameStr(collform->collname))) :
+							 errhint("Install another version of ICU and select it using "
+									 "default_icu_libary_version, "
+									 "or rebuild all objects affect by this collation and run "
+									 "ALTER COLLATION %s REFRESH VERSION, "
+									 "or build PostgreSQL with the right library version.",
+									 quote_qualified_identifier(get_namespace_name(collform->collnamespace),
+																NameStr(collform->collname)))));
+				}
+				else
+				{
+					ereport(WARNING,
+							(errmsg("collation \"%s\" has version mismatch",
+									NameStr(collform->collname)),
+							 errdetail("The collation in the database was created using version %s, "
+									   "but the operating system provides version %s.",
+									   collversionstr, actual_versionstr),
+							 errhint("Rebuild all objects affected by this collation and run "
+									 "ALTER COLLATION %s REFRESH VERSION, "
+									 "or build PostgreSQL with the right library version.",
+									 quote_qualified_identifier(get_namespace_name(collform->collnamespace),
+																NameStr(collform->collname)))));
+				}
+			}
 		}
 
 		ReleaseSysCache(tp);
@@ -1651,21 +2021,23 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 #ifdef USE_ICU
 	if (collprovider == COLLPROVIDER_ICU)
 	{
+		pg_icu_library *lib;
 		UCollator  *collator;
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
 
+		lib = get_icu_library_for_collation(collcollate, &collcollate);
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collcollate, &status);
+		collator = lib->open(collcollate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collcollate, u_errorName(status))));
-		ucol_getVersion(collator, versioninfo);
-		ucol_close(collator);
+							collcollate, lib->errorName(status))));
+		lib->getVersion(collator, versioninfo);
+		lib->close(collator);
 
-		u_versionToString(versioninfo, buf);
+		lib->versionToString(versioninfo, buf);
 		collversion = pstrdup(buf);
 	}
 	else
@@ -1733,6 +2105,33 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 
 
 #ifdef USE_ICU
+
+/*
+ * Given a major version number, look up that library and ask it for the
+ * complete version string.
+ */
+Datum
+pg_icu_library_version(PG_FUNCTION_ARGS)
+{
+#ifdef USE_ICU
+	int			major_version;
+	pg_icu_library *lib;
+	UVersionInfo versioninfo;
+	char		buf[U_MAX_VERSION_STRING_LENGTH];
+
+	major_version = PG_GETARG_INT32(0);
+	if (major_version <= 0)
+		major_version = U_ICU_VERSION_MAJOR_NUM;
+
+	lib = get_icu_library(major_version);
+	lib->getLibraryVersion(versioninfo);
+	lib->versionToString(versioninfo, buf);
+	PG_RETURN_TEXT_P(cstring_to_text(buf));
+#else
+	PG_RETURN_NULL();
+#endif
+}
+
 /*
  * Converter object for converting between ICU's UChar strings and C strings
  * in database encoding.  Since the database encoding doesn't change, we only
@@ -1954,19 +2353,21 @@ void
 check_icu_locale(const char *icu_locale)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_icu_library_for_collation(icu_locale, &icu_locale);
 	status = U_ZERO_ERROR;
-	collator = ucol_open(icu_locale, &status);
+	collator = lib->open(icu_locale, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						icu_locale, u_errorName(status))));
+						icu_locale, lib->errorName(status))));
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
+	if (lib->major_version < 54)
 		icu_set_collation_attributes(collator, icu_locale);
-	ucol_close(collator);
+	lib->close(collator);
 #else
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index 68e2e6f7a7..e0c86870e0 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -1026,11 +1026,11 @@ hashbpchar(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, keydata, keylen);
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -1087,11 +1087,11 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c5e7ee7ca2..cf891a5654 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1667,13 +1667,14 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					UErrorCode	status;
 
 					status = U_ZERO_ERROR;
-					result = ucol_strcollUTF8(mylocale->info.icu.ucol,
-											  arg1, len1,
-											  arg2, len2,
-											  &status);
+					result = PG_ICU_LIB(mylocale)->strcollUTF8(PG_ICU_COL(mylocale),
+															   arg1, len1,
+															   arg2, len2,
+															   &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
-								(errmsg("collation failed: %s", u_errorName(status))));
+								(errmsg("collation failed: %s",
+										PG_ICU_LIB(mylocale)->errorName(status))));
 				}
 				else
 #endif
@@ -1686,9 +1687,9 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					ulen1 = icu_to_uchar(&uchar1, arg1, len1);
 					ulen2 = icu_to_uchar(&uchar2, arg2, len2);
 
-					result = ucol_strcoll(mylocale->info.icu.ucol,
-										  uchar1, ulen1,
-										  uchar2, ulen2);
+					result = PG_ICU_LIB(mylocale)->strcoll(PG_ICU_COL(mylocale),
+														   uchar1, ulen1,
+														   uchar2, ulen2);
 
 					pfree(uchar1);
 					pfree(uchar2);
@@ -2388,13 +2389,14 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				UErrorCode	status;
 
 				status = U_ZERO_ERROR;
-				result = ucol_strcollUTF8(sss->locale->info.icu.ucol,
-										  a1p, len1,
-										  a2p, len2,
-										  &status);
+				result = PG_ICU_LIB(sss->locale)->strcollUTF8(PG_ICU_COL(sss->locale),
+															  a1p, len1,
+															  a2p, len2,
+															  &status);
 				if (U_FAILURE(status))
 					ereport(ERROR,
-							(errmsg("collation failed: %s", u_errorName(status))));
+							(errmsg("collation failed: %s",
+									PG_ICU_LIB(sss->locale)->errorName(status))));
 			}
 			else
 #endif
@@ -2407,9 +2409,9 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				ulen1 = icu_to_uchar(&uchar1, a1p, len1);
 				ulen2 = icu_to_uchar(&uchar2, a2p, len2);
 
-				result = ucol_strcoll(sss->locale->info.icu.ucol,
-									  uchar1, ulen1,
-									  uchar2, ulen2);
+				result = PG_ICU_LIB(sss->locale)->strcoll(PG_ICU_COL(sss->locale),
+														  uchar1, ulen1,
+														  uchar2, ulen2);
 
 				pfree(uchar1);
 				pfree(uchar2);
@@ -2569,24 +2571,24 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 					uint32_t	state[2];
 					UErrorCode	status;
 
-					uiter_setUTF8(&iter, sss->buf1, len);
+					PG_ICU_LIB(sss->locale)->setUTF8(&iter, sss->buf1, len);
 					state[0] = state[1] = 0;	/* won't need that again */
 					status = U_ZERO_ERROR;
-					bsize = ucol_nextSortKeyPart(sss->locale->info.icu.ucol,
-												 &iter,
-												 state,
-												 (uint8_t *) sss->buf2,
-												 Min(sizeof(Datum), sss->buflen2),
-												 &status);
+					bsize = PG_ICU_LIB(sss->locale)->nextSortKeyPart(PG_ICU_COL(sss->locale),
+																	 &iter,
+																	 state,
+																	 (uint8_t *) sss->buf2,
+																	 Min(sizeof(Datum), sss->buflen2),
+																	 &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
 								(errmsg("sort key generation failed: %s",
-										u_errorName(status))));
+										PG_ICU_LIB(sss->locale)->errorName(status))));
 				}
 				else
-					bsize = ucol_getSortKey(sss->locale->info.icu.ucol,
-											uchar, ulen,
-											(uint8_t *) sss->buf2, sss->buflen2);
+					bsize = PG_ICU_LIB(sss->locale)->getSortKey(PG_ICU_COL(sss->locale),
+																uchar, ulen,
+																(uint8_t *) sss->buf2, sss->buflen2);
 			}
 			else
 #endif
diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c
index a990c833c5..d18aa7a3df 100644
--- a/src/backend/utils/init/postinit.c
+++ b/src/backend/utils/init/postinit.c
@@ -449,26 +449,52 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 	{
 		char	   *actual_versionstr;
 		char	   *collversionstr;
+		char	   *locale;
 
 		collversionstr = TextDatumGetCString(datum);
+		locale = dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : collate;
 
-		actual_versionstr = get_collation_actual_version(dbform->datlocprovider, dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : collate);
+		actual_versionstr = get_collation_actual_version(dbform->datlocprovider, locale);
 		if (!actual_versionstr)
 			/* should not happen */
 			elog(WARNING,
 				 "database \"%s\" has no actual collation version, but a version was recorded",
 				 name);
 		else if (strcmp(actual_versionstr, collversionstr) != 0)
-			ereport(WARNING,
-					(errmsg("database \"%s\" has a collation version mismatch",
-							name),
-					 errdetail("The database was created using collation version %s, "
-							   "but the operating system provides version %s.",
-							   collversionstr, actual_versionstr),
-					 errhint("Rebuild all objects in this database that use the default collation and run "
-							 "ALTER DATABASE %s REFRESH COLLATION VERSION, "
-							 "or build PostgreSQL with the right library version.",
-							 quote_identifier(name))));
+		{
+			if (dbform->datlocprovider == COLLPROVIDER_ICU)
+			{
+				ereport(WARNING,
+						(errmsg("database \"%s\" has a collation version mismatch",
+								name),
+						 errdetail("The database was created using collation version %s, "
+								   "but the ICU library provides version %s.",
+								   collversionstr, actual_versionstr),
+						 strchr(locale, ':') != NULL ?
+						 errhint("Rebuild all objects in this database that use the default collation and run "
+								 "ALTER DATABASE %s REFRESH COLLATION VERSION, "
+								 "or build PostgreSQL with the right library version.",
+								 quote_identifier(name)) :
+						 errhint("Install another version of ICU and select it using default_icu_library_verison, or "
+								 "rebuild all objects in this database that use the default collation and run "
+								 "ALTER DATABASE %s REFRESH COLLATION VERSION, "
+								 "or build PostgreSQL with the right library version.",
+								 quote_identifier(name))));
+			}
+			else
+			{
+				ereport(WARNING,
+						(errmsg("database \"%s\" has a collation version mismatch",
+								name),
+						 errdetail("The database was created using collation version %s, "
+								   "but the operating system provides version %s.",
+								   collversionstr, actual_versionstr),
+						 errhint("Rebuild all objects in this database that use the default collation and run "
+								 "ALTER DATABASE %s REFRESH COLLATION VERSION, "
+								 "or build PostgreSQL with the right library version.",
+								 quote_identifier(name))));
+			}
+		}
 	}
 
 	/* Make the locale settings visible as GUC variables, too */
diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c
index 836b49484a..25e905ce8b 100644
--- a/src/backend/utils/misc/guc_tables.c
+++ b/src/backend/utils/misc/guc_tables.c
@@ -2937,6 +2937,20 @@ struct config_int ConfigureNamesInt[] =
 		check_max_worker_processes, NULL, NULL
 	},
 
+	{
+		{"default_icu_library_version",
+			PGC_SUSET,
+			COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Default major version of ICU library to use for collations if not specified."),
+			NULL
+		},
+		&default_icu_library_version,
+		0,
+		0,
+		1000,
+		NULL, NULL, NULL
+	},
+
 	{
 		{"max_logical_replication_workers",
 			PGC_POSTMASTER,
@@ -3920,6 +3934,20 @@ struct config_string ConfigureNamesString[] =
 		NULL, NULL, NULL
 	},
 
+	{
+		{"icu_library_path", PGC_SUSET, COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Sets the path for dynamically loadable ICU libraries."),
+			gettext_noop("If versions of ICU other than the one that "
+						 "PostgreSQL is linked against are needed, they will "
+						 "be opened from this directory.  If empty, the "
+						 "system linker search path will be used."),
+			GUC_SUPERUSER_ONLY
+		},
+		&icu_library_path,
+		"",
+		NULL, NULL, NULL
+	},
+
 	{
 		{"krb_server_keyfile", PGC_SIGHUP, CONN_AUTH_AUTH,
 			gettext_noop("Sets the location of the Kerberos server key file."),
diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample
index 868d21c351..2713c92124 100644
--- a/src/backend/utils/misc/postgresql.conf.sample
+++ b/src/backend/utils/misc/postgresql.conf.sample
@@ -727,6 +727,11 @@
 #lc_numeric = 'C'			# locale for number formatting
 #lc_time = 'C'				# locale for time formatting
 
+#default_icu_library_version = 0	# default major version of ICU library
+					# (0 for the linked version)
+#icu_library_path = ''			# path for dynamically loaded ICU
+					# libraries
+
 # default configuration for text search
 #default_text_search_config = 'pg_catalog.simple'
 
diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat
index 9dbe9ec801..a76eb6c94b 100644
--- a/src/include/catalog/pg_proc.dat
+++ b/src/include/catalog/pg_proc.dat
@@ -11707,6 +11707,9 @@
   proname => 'pg_database_collation_actual_version', procost => '100',
   provolatile => 'v', prorettype => 'text', proargtypes => 'oid',
   prosrc => 'pg_database_collation_actual_version' },
+{ oid => '8888', descr => 'get ICU library version string',
+  proname => 'pg_icu_library_version', provolatile => 'v', prorettype => 'text',
+  proargtypes => 'int4', prosrc => 'pg_icu_library_version' },
 
 # system management/monitoring related functions
 { oid => '3353', descr => 'list files in the log directory',
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index a875942123..c52fe0c7df 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -17,6 +17,7 @@
 #endif
 #ifdef USE_ICU
 #include <unicode/ucol.h>
+#include <unicode/ubrk.h>
 #endif
 
 #ifdef USE_ICU
@@ -40,6 +41,8 @@ extern PGDLLIMPORT char *locale_messages;
 extern PGDLLIMPORT char *locale_monetary;
 extern PGDLLIMPORT char *locale_numeric;
 extern PGDLLIMPORT char *locale_time;
+extern PGDLLIMPORT char *icu_library_path;
+extern PGDLLIMPORT int default_icu_library_version;
 
 /* lc_time localization cache */
 extern PGDLLIMPORT char *localized_abbrev_days[];
@@ -63,6 +66,72 @@ extern struct lconv *PGLC_localeconv(void);
 
 extern void cache_locale_time(void);
 
+#ifdef USE_ICU
+
+/*
+ * An ICU library version that we're either linked against or have loaded at
+ * runtime.
+ */
+typedef struct pg_icu_library
+{
+	int			major_version;
+	void	   *libicui18n_handle;
+	void	   *libicuuc_handle;
+	void		(*getLibraryVersion) (UVersionInfo info);
+	UCollator  *(*open) (const char *loc, UErrorCode *status);
+	void		(*close) (UCollator *coll);
+	void		(*getVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*versionToString) (const UVersionInfo versionArray,
+									char *versionString);
+				UCollationResult(*strcoll) (const UCollator *coll,
+											const UChar *source,
+											int32_t sourceLength,
+											const UChar *target,
+											int32_t targetLength);
+				UCollationResult(*strcollUTF8) (const UCollator *coll,
+												const char *source,
+												int32_t sourceLength,
+												const char *target,
+												int32_t targetLength,
+												UErrorCode *status);
+	int32_t		(*getSortKey) (const UCollator *coll,
+							   const UChar *source,
+							   int32_t sourceLength,
+							   uint8_t *result,
+							   int32_t resultLength);
+	int32_t		(*nextSortKeyPart) (const UCollator *coll,
+									UCharIterator *iter,
+									uint32_t state[2],
+									uint8_t *dest,
+									int32_t count,
+									UErrorCode *status);
+	void		(*setUTF8) (UCharIterator *iter,
+							const char *s,
+							int32_t length);
+	const char *(*errorName) (UErrorCode code);
+	int32_t		(*strToUpper) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToLower) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToTitle) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UBreakIterator *titleIter,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	struct pg_icu_library *next;
+} pg_icu_library;
+
+#endif
 
 /*
  * We define our own wrapper around locale_t so we can keep the same
@@ -84,12 +153,18 @@ struct pg_locale_struct
 		{
 			const char *locale;
 			UCollator  *ucol;
+			pg_icu_library *lib;
 		}			icu;
 #endif
 		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
 	}			info;
 };
 
+#ifdef USE_ICU
+#define PG_ICU_LIB(x) ((x)->info.icu.lib)
+#define PG_ICU_COL(x) ((x)->info.icu.ucol)
+#endif
+
 typedef struct pg_locale_struct *pg_locale_t;
 
 extern PGDLLIMPORT struct pg_locale_struct default_locale;
diff --git a/src/test/icu/meson.build b/src/test/icu/meson.build
index 5a4f53f37f..ac2672190e 100644
--- a/src/test/icu/meson.build
+++ b/src/test/icu/meson.build
@@ -5,6 +5,7 @@ tests += {
   'tap': {
     'tests': [
       't/010_database.pl',
+      't/020_multiversion.pl',
     ],
     'env': {'with_icu': icu.found() ? 'yes' : 'no'},
   },
diff --git a/src/test/icu/t/020_multiversion.pl b/src/test/icu/t/020_multiversion.pl
new file mode 100644
index 0000000000..52408deb59
--- /dev/null
+++ b/src/test/icu/t/020_multiversion.pl
@@ -0,0 +1,203 @@
+# Copyright (c) 2022, PostgreSQL Global Development Group
+
+# This test requires a second major version of ICU installed in the usual
+# system library search path.  That is, not the one PostgreSQL was linked
+# against.  It also assumes that ucol_getVersion() for locale "en" will change
+# between the two library versions.
+
+use strict;
+use warnings;
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Test::More;
+
+if ($ENV{with_icu} ne 'yes')
+{
+	plan skip_all => 'ICU not supported by this build';
+}
+
+if (!($ENV{PG_TEST_EXTRA} =~ /\bicu=([0-9]+)\b/))
+{
+	plan skip_all => 'PG_TEST_EXTRA not configured to test an alternative ICU library version';
+}
+my $alt_major_version = $1;
+
+my $node1 = PostgreSQL::Test::Cluster->new('node1');
+$node1->init;
+$node1->start;
+
+my $linked_major_version = $node1->safe_psql('postgres', 'select pg_icu_library_version(-1)::decimal::int');
+
+print "linked_major_version = $linked_major_version\n";
+print "alt_major_version = $alt_major_version\n";
+
+if ($alt_major_version ge $linked_major_version)
+{
+	BAIL_OUT("can't run multi-version tests because ICU major version selected via PG_TEST_EXTRA is not lower than the major version the executable is linked against ($linked_major_version)");
+}
+
+# Sanity check that when we load a library, its u_getVersion() function tells
+# us it has the major version we expect.  The result is a string eg "71.1", so
+# we get the major part by casting.
+is($node1->safe_psql('postgres', "select pg_icu_library_version($alt_major_version)::decimal::int"),
+	$alt_major_version,
+	"alt library reports expected major version");
+
+sub set_default_icu_library_version
+{
+	my $major_version = shift;
+	$node1->safe_psql('postgres', "alter system set default_icu_library_version = $major_version; select pg_reload_conf()");
+}
+
+my $ret;
+my $stderr;
+
+# Create a collation that doesn't specify the ICU version to use.  Which
+# library we load depends on the GUC default_icu_library_version.  Here it uses
+# the linked version because it's set to 0 (default value in a new cluster).
+set_default_icu_library_version(0);
+$node1->safe_psql('postgres', "create collation c1 (provider=icu, locale='en')");
+
+# No warning by default.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c1", stderr => \$stderr);
+is($ret, 0, "can use collation");
+unlike($stderr, qr/WARNING/, "no warning for default");
+
+# No warning if we explicitly select the linked version.
+set_default_icu_library_version($linked_major_version);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c1", stderr => \$stderr);
+unlike($stderr, qr/WARNING/, "no warning for explicit match");
+
+# If we use a different major version explicitly, we get a warning that
+# includes a hint that we might be able to install and select a different ICU
+# version.
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c1", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING/, "warning for incorrect major version");
+like($stderr, qr/HINT:  Install another version of ICU/, "warning suggests installing another ICU version");
+
+# Create a collation using the alt version without specifying it explicitly.
+# This simulates a collation that was created by a different build linked
+# against an older ICU.
+$node1->safe_psql('postgres', "create collation c2 (provider=icu, locale='en')");
+
+# Warning if we try to use it with default setttings.
+set_default_icu_library_version(0);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING/, "warning for incorrect major version");
+like($stderr, qr/HINT:  Install another version of ICU/, "warning suggests installing another ICU version");
+
+# No warning if we explicitly activate the alt version.
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning for explicit match");
+
+# Refresh the version...  this will update it from the linked version (or
+# whatever default_icu_library_version points to, here it's 0 and thus the
+# linked version), because c2 is not explicitly pinned to an ICU major version.
+set_default_icu_library_version(0);
+$ret = $node1->psql('postgres', "alter collation c2 refresh version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+
+# Now no warning.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "warning has gone away after refresh");
+
+# Create a collation that is pinned to a specific version of ICU.
+$node1->safe_psql('postgres', "create collation c3 (provider=icu, locale='$alt_major_version:en')");
+
+# No warnings expected, no matter what default_icu_library_version says, because
+# we always load that exact library.
+set_default_icu_library_version($linked_major_version);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c3", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning for explicit lib");
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c3", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning for explicit lib");
+set_default_icu_library_version(0);
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c3", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning for explicit lib");
+
+# Similar tests using the database default.
+
+set_default_icu_library_version(0);
+$node1->safe_psql('postgres', "create database db2 locale_provider = icu template = template0 icu_locale = 'en'");
+
+# No warning.
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# Warning when you log into the database.
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING/, "warning for incorrect major version");
+like($stderr, qr/HINT:  Install another version of ICU/, "warning suggests installing another ICU version");
+
+# One way to clear the warning is to REFRESH.
+$ret = $node1->psql('postgres', "alter database db2 refresh collation version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+
+# Now the warning is gone.
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# Now we go back to using the linked version, and we'll see the warning again.
+# Perhaps this case simulates the most likely real-world experience, when
+# moving to a new OS that has PostgreSQL packages linked against a later ICU
+# version, using all defaults.
+set_default_icu_library_version(0);
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING/, "warning for incorrect major version");
+like($stderr, qr/HINT:  Install another version of ICU/, "warning suggests installing another ICU version");
+
+# Option 1 is to get rid of the warning by installing the library and setting
+# the GUC.
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning after setting GUC");
+
+# Option 2 is to rebuild indexes etc and use REFRESH.
+set_default_icu_library_version(0);
+$ret = $node1->psql('postgres', "alter database db2 refresh collation version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning after refresh");
+
+# None of this applies if you explicitly pinned your database to an specific
+# ICU major version in the first place, so we ignore the GUC.
+set_default_icu_library_version(0);
+$node1->safe_psql('postgres', "create database db3 locale_provider = icu template = template0 icu_locale = '$alt_major_version:en'");
+
+# No warning with all GUC settings.
+set_default_icu_library_version($alt_major_version);
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning with pinned library version");
+set_default_icu_library_version($linked_major_version);
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning with pinned library version");
+set_default_icu_library_version(0);
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning with pinned library version");
+
+$node1->stop;
+
+done_testing();
diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list
index f8302f1ed1..12c11f1586 100644
--- a/src/tools/pgindent/typedefs.list
+++ b/src/tools/pgindent/typedefs.list
@@ -1101,6 +1101,7 @@ HeapTupleTableSlot
 HistControl
 HotStandbyState
 I32
+ICU_Convert_BI_Func
 ICU_Convert_Func
 ID
 INFIX
@@ -2854,6 +2855,7 @@ TypeName
 U
 U32
 U8
+UBreakIterator
 UChar
 UCharIterator
 UColAttribute
@@ -3484,6 +3486,7 @@ pg_funcptr_t
 pg_gssinfo
 pg_hmac_ctx
 pg_hmac_errno
+pg_icu_library
 pg_int64
 pg_local_to_utf_combined
 pg_locale_t
-- 
2.30.2



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-18 19:16  Thomas Munro <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-18 19:16 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sat, Nov 19, 2022 at 7:38 AM Thomas Munro <[email protected]> wrote:
> On Tue, Nov 15, 2022 at 1:55 PM Jeff Davis <[email protected]> wrote:
> > I realize your patch is experimental, but when there is a better
> > consensus on the approach, we should consider adding declarative syntax
> > such as:
> >
> >    CREATE COLLATION (or LOCALE?) PROVIDER icu67
> >      TYPE icu VERSION '67' AS '/path/to/icui18n.so.67';
> >
> > It will offer more opportunities to catch errors early and offer better
> > error messages. It would also enable it to function if the library is
> > built with --disable-renaming (though we'd have to trust the user).
>
> Earlier in this and other threads, we wondered if each ICU major version should
> be a separate provider, which is what you're showing there, or should be an
> independent property of an individual COLLATION, which is what v6 did with
> '63:en' and what Peter suggested I make more formal with CREATE COLLATION foo
> (..., ICU_VERSION=63).  I actually started out thinking we'd have multiple
> providers, but I couldn't really think of any advantage, and I think it makes
> some upgrade scenarios more painful.  Can you elaborate on why you'd want
> that model?

Hmm, thinking some more about this... I said the above thinking that
you couldn't change a provider after creating a database/collation.
But what if you could?

1.  CREATE DATABASE x LOCALE_PROVIDER=icu ...;
2.  Some time later after an upgrade, my postgres binary is linked
against a new ICU version and I start seeing warnings.
3.  ALTER DATABASE x LOCALE_PROVIDER=icu63;

I suppose you shouldn't be allowed to change libc -> icu, but you
could change icu - > icuXXX, or I guess icuXXX -> icuXXX.

What if you didn't have to manually manage the set of available
providers with DDL like you showed, but we just automatically
supported "icu" (= the linked ICU, whatever it might be), and icu50 up
to icuXXX where XXX is the linked ICU's version?  We can encode those
values + libc as an int, to replace the existing char the represents
providers in catalogues.

That's basically just a different way of encoding the same information
that Peter was suggesting I put in a new catalogue attribute.  How do
you like that bikeshed colour?





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-22 06:34  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-22 06:34 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers


On Sat, 2022-10-22 at 14:22 +1300, Thomas Munro wrote:
> Problem 2:  If ICU 67 ever decides to report a different version for
> a
> given collation (would it ever do that?  I don't expect so, but ...),
> we'd be unable to open the collation with the search-by-collversion
> design, and potentially the database.  What is a user supposed to do
> then?  Presumably our error/hint for that would be "please insert the
> correct ICU library into drive A", but now there is no correct
> library

Let's say that Postgres is compiled against version 67.X, and the
sysadmin upgrades the ICU package to 67.Y, which reports a different
collation version for some locale.

Your current patch makes this impossible for the administrator to fix,
because there's no way to have two different libraries loaded with the
same major version number, so it will always pick the compiled-in ICU.
The user will be forced to accept the new version of the collation, see
WARNINGs in their logs, and possibly corrupt their indexes.

Search-by-collversion would still be frustrating for the admin, but at
least it would be possible to fix by compiling their own 67.X and
asking Postgres to search that library, too. We could make it slightly
more friendly by having an error that reports the libraries searched
and the collation versions found, if none of the versions match. We can
have a GUC that controls whether a failure to find the right version is
a WARNING or an ERROR.

On Sat, 2022-11-19 at 07:38 +1300, Thomas Munro wrote:
> >   * We'll need some clearer instructions on how to build/install
> > extra
> > ICU versions that might not be provided by the distribution
> > packaging.
> > For instance, I got a cryptic error until I used --enable-rpath,
> > which
> > might not be obvious to all users.
> 
> Suggestions welcome.  No docs at all yet...

I tried to write up some docs. It's hard to explain why we are exposing
to the user the collation version and the library version in these
different ways, and what effects they have.

The current patch feels like it hasn't decided whether the collation
version is ucol_getVersion() (collversion) or u_getVersion() (library
version). The collversion is more prominent in the UI (with its own
syntax), yet it's just a cross-check for whether to issue a WARNING or
not; while the library version is hidden in the locale field and it
actually decides which symbol is called.

> 
> 
> Yeah.  I just don't like the way it *appears* to be doing something
> clever, but
> it doesn't solve any fundamental problem at all because the
> collversion
> information is under human control and so it's really doing something
> stupid.

I assume by "human control" you mean "ALTER COLLATION ... REFRESH
VERSION". I agree that relying on the admin's declaration is dubious,
especially when we provide no good advice on how to actually do that
safely.

But I don't see what using the library version instead buys us here,
except that library version is part of the LOCALE, and there's no ALTER
command for that. You could just as easily deprecate/eliminate the
ALTER COLLATION REFRESH VERSION, and then say that the collversion is
out of human control, too.

By introducing multiple libraries, I think we need to change that
syntax anyway, to be something like:

   ALTER COLLATION ... SET VERSION TO '...'

or even:

   ALTER COLLATION ... FORCE VERSION TO '...'

> Hence desire to build something that at least admits that it's
> primitive and
> just gives you some controls, in a first version.

Using either the library version or the collation version seems
reasonably simple to me. But from a documentation and usability
standpoint, the way they are currently mixed seems confusing.



-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-23 05:08  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 2 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-23 05:08 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 22, 2022 at 7:34 PM Jeff Davis <[email protected]> wrote:
> On Sat, 2022-10-22 at 14:22 +1300, Thomas Munro wrote:
> > Problem 2:  If ICU 67 ever decides to report a different version for
> > a
> > given collation (would it ever do that?  I don't expect so, but ...),
> > we'd be unable to open the collation with the search-by-collversion
> > design, and potentially the database.  What is a user supposed to do
> > then?  Presumably our error/hint for that would be "please insert the
> > correct ICU library into drive A", but now there is no correct
> > library
>
> Let's say that Postgres is compiled against version 67.X, and the
> sysadmin upgrades the ICU package to 67.Y, which reports a different
> collation version for some locale.
>
> Your current patch makes this impossible for the administrator to fix,
> because there's no way to have two different libraries loaded with the
> same major version number, so it will always pick the compiled-in ICU.
> The user will be forced to accept the new version of the collation, see
> WARNINGs in their logs, and possibly corrupt their indexes.

They could probably also 'pin' the older minor version package using
their package manager (= downgrade) until they're ready to upgrade and
use REFRESH VERSION to certify that they've rebuilt everything
relevant or are OK with risks.  Not pretty I admit, but I think the
end result is about the same for search-for-collversion, because I
imagine that (1) the default behaviour on failure to search would
likely be to use the linked library instead and WARN about
[dat]collversion mismatch, so far the same, and (2) the set of people
who would really be prepared to compile their own copy of 67.X instead
of downgrading or REFRESHing (with or without rebuilding) is
vanishingly small.

Two questions I wondered about:

1.  *Do* they change ucol_getVersion() values in minor releases?  I
tried to find a written policy on that.
https://icu.unicode.org/processes is not encouraging: it gives the
example of a "third digit in an official release number" [changing]
because a CLDR change was incorporated.  Hrmph.  But that's clearly
not even the modern ICU versioning system (it made a change a bit like
ours in 49, making the first number only major, so maybe that "third"
number is now the second number, AKA minor version), and also that's a
CLDR minor version change; is CLDR minor even in the recipe for
ucol_getVersion()?  Even without data changes, I guess that bug fixes
could apply to the UCA logic, and I assume that UCA logic is included
in it.  Hmm.

A non-hypothetical example of a CLDR change within an ICU major
version that I've been able to find is:

https://cldr.unicode.org/index/downloads/cldr-38

Here we see that CLDR had a minor version bump 38 -> 38.1, "a very
small number of incremental additions to version 38 to address the
specific bugs listed in Δ38.1", and was included in ICU 68.2.  Being a
minor ICU release 68.1 -> 68.2, perhaps you could finish up running
that just with a regular upgrade on typical distros (not a major OS
upgrade), and since PostgreSQL would normally be linked against eg
.68, not .68.1, it'd start using it at the next cluster start when
that symlink is updated to point to .68.2.  As it happens, if you
follow the documentation links to see what actually changed in that
particular pair of CLDR+ICU minor releases, it's timezones and locale
stuff other than collations, so wouldn't affect us.  Can we find a
chapter and verse that says that ICU would only ever move to a new
CLDR in a minor release, and CLDR would never change order of
pre-existing code points in a minor release?

It might be interesting to see if
https://github.com/unicode-org/icu/tree/release-68-1 and
https://github.com/unicode-org/icu/tree/release-68-2 report a
different ucol_getVersion() for any locale, but not conclusive if it
doesn't; it might be because something in the version pipeline knew
that particular CLDR change didn't affect collators...

This speculation feels pretty useless.  Maybe we should go and read
the code or ask an ICU expert, but I'm not against making it
theoretically possible to access two different minor versions at once,
just to cover all the bases for future-proofing.

2.  Would package managers ever allow two minor versions to be
installed at once?  I highly doubt it; they're probably more
interested in ABI stability so that dependent packages work when
bugfixes are shipped, and that's certainly nailed down at the major
version level.  It'd probably be a case of having to compile it
yourself, which seems unlikely to me in the real world.  That's why I
left minor version out of earlier patches, but I'm OK with changing
that.

As for how, I think that depends on our modelling decision (see below).

> Search-by-collversion would still be frustrating for the admin, but at
> least it would be possible to fix by compiling their own 67.X and
> asking Postgres to search that library, too. We could make it slightly
> more friendly by having an error that reports the libraries searched
> and the collation versions found, if none of the versions match. We can
> have a GUC that controls whether a failure to find the right version is
> a WARNING or an ERROR.

Good ideas.

> I tried to write up some docs. It's hard to explain why we are exposing
> to the user the collation version and the library version in these
> different ways, and what effects they have.

Always a good test: see how crazy it sounds when translated to user speak.

> The current patch feels like it hasn't decided whether the collation
> version is ucol_getVersion() (collversion) or u_getVersion() (library
> version). The collversion is more prominent in the UI (with its own
> syntax), yet it's just a cross-check for whether to issue a WARNING or
> not; while the library version is hidden in the locale field and it
> actually decides which symbol is called.

Yeah.  I agree that it sucks to have two kinds of versions flying
around in the user's mind.

> > Yeah.  I just don't like the way it *appears* to be doing something
> > clever, but
> > it doesn't solve any fundamental problem at all because the
> > collversion
> > information is under human control and so it's really doing something
> > stupid.
>
> I assume by "human control" you mean "ALTER COLLATION ... REFRESH
> VERSION". I agree that relying on the admin's declaration is dubious,
> especially when we provide no good advice on how to actually do that
> safely.
>
> But I don't see what using the library version instead buys us here,
> except that library version is part of the LOCALE, and there's no ALTER
> command for that. You could just as easily deprecate/eliminate the
> ALTER COLLATION REFRESH VERSION, and then say that the collversion is
> out of human control, too.
>
> By introducing multiple libraries, I think we need to change that
> syntax anyway, to be something like:
>
>    ALTER COLLATION ... SET VERSION TO '...'
>
> or even:
>
>    ALTER COLLATION ... FORCE VERSION TO '...'

OK.  Time for a new list of the various models we've discussed so far:

1.  search-by-collversion:  We introduce no new "library version"
concept to COLLATION and DATABASE object and little or no new syntax.
Whenever opening a collation or database, the system will search some
candidate list of ICU libraries to try to find the one that agrees
with [dat]collversion.  When creating a new collation or database, the
system will select one (probably the linked one unless you override
somehow) and record ucol_getVersion() in [dat]collversion.  When
searching, it might fail to find a suitable library and ereport; to
fix that, it is the admin's job to somehow expand the set of candidate
libraries.  In such a failure case, perhaps it would fall back to
using some default library version (probably the one that is linked,
overridable by GUC?), with a WARNING (unless you turned on ERRORs),
and if you want to shut it up without supplying the right candidate
library, you can still fall back to the REFRESH VERSION hammer (or
maybe that should indeed called FORCE to make it clearer that it's not
a harmless operation where the system holds your hand, you're actually
certifying that you have rebuilt indexes and you know what you're
doing).

The set of candidate versions could perhaps be provided with
extra_icu_library_versions=63,71 OR =63.1,63.2 strings, at least on
Unix systems following the traditional symlink conventions.
Remembering that a typical Unixoid system should have libraries and
symlinks like:

  libicui18n.a
  libicui18n.so -> libicui18n.so.71.1
  libicui18n.so.63 -> libicui18n.so.63.1
  libicui18n.so.63.1
  libicui18n.so.67 -> libicui18n.so.67.1
  libicui18n.so.67.1
  libicui18n.so.71 -> libicui18n.so.71.1
  libicui18n.so.71.1

The reason I prefer major[.minor] strings over whole library names is
that we need to dlopen two of them so it's a little easier to build
them from those parts than have to supply both names.  The reason I
prefer to keep allowing major-only versions to be listed is that it's
good to have the option to just follow minor upgrades automatically.
Or I guess you could make something that can automatically search a
whole directory (which directory?) to find all the suitably named
libraries so you don't ever have to mention versions manually (if you
want "apt-get install libicu72" to be enough with no GUC change
needed) -- is that too weird?

Perhaps we could write functions that can show you the available
versions to demystify the searching mechanism slightly and show how
various numbers relate, something like (warning: I made up numbers for
illustration, they are wrong!):

  SELECT * FROM pg_available_icu_libraries()

  icu_version unicode_version uca_version  cldr_version
  67.1        14.0            3.1          38.0
  71.1        15.0            4.0          42.0

  SELECT * FROM pg_available_icu_collation_versions('en')

  icu_version collation_version
  67.1        142.42
  71.1        153.112

2.  lib-version-in-providers: We introduce a separate provider value
for each ICU version, for example ICU63, plus an unversioned ICU like
today.  The collversion column is used only for warnings.  Warnings
are expected when you used the unversioned ICU provider and upgrade to
a binary linked to a later library.  You can clear the warnings by
doing ALTER COLLATION/DATABASE SET [LOCALE_]PROVIDER = ICU63, or with
the REFRESH VERSION hammer.

Not sure how you fit minor versions into that, if we want to support
those.  Maybe ICU means "whatever is linked", ICU63 means "whatever
libicui18n.so.63 points to" and ICU63_1 means libicu18n.so.63.1,
something like that, so the user can choose from three levels of
specificity.

3.  lib-version-in-attributes: We introduce daticuversion (alongside
datcollversion) and collicuversion (alongside collversion).  Similar
to the above, but it's a separate property and the provider is always
ICU.  New syntax for CREATE/ALTER COLLATION/DATABASE to set and change
ICU_VERSION.

4.  lib-version-in-locale:  "63:en" from earlier versions.  That was
mostly a strawman proposal to avoid getting bogged down in
syntax/catalogue/model change discussions while trying to prove that
dlopen would even work.  It doesn't sound like anyone really likes
this.

5.  lib-version-in-collversion:  We didn't explicitly discuss this
before, but you hinted at it: we could just use u_getVersion() in
[dat]collversion.  I haven't analysed this much but I don't think it
has a very nice upgrade path from PG15, and it forces you to decide
whether to store just the major version and not even notice when the
(unstored) minor version changes, or store major.minor and
complain/break down when routine minor upgrades happen.  It is a
logical possibility though, once you decide you only want one kind of
version in the system.

I'm willing to update the patch to try one of these out so we can kick
the tyres some more, but I'll wait to see if we can get some consensus
on the way forward.  Despite my initial reactions, I'm willing to try
out the search-by-collversion concept if others are keen on it.  The
example I worked through in the first paragraph of this email helped
me warm to it a little, and with the observability functions I showed
you might have a chance of figuring out what's going on in some edge
cases.  Any other ideas, or votes for these ideas?





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-24 02:07  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 2 replies; 57+ messages in thread

From: Jeff Davis @ 2022-11-24 02:07 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-23 at 18:08 +1300, Thomas Munro wrote:

> (1) the default behaviour on failure to search would
> likely be to use the linked library instead and WARN about
> [dat]collversion mismatch, so far the same, and 

Agreed.

> (2) the set of people
> who would really be prepared to compile their own copy of 67.X
> instead
> of downgrading or REFRESHing (with or without rebuilding) is
> vanishingly small.

The set of people prepared to do so is probably small. But the set of
people who will do it (prepared or not) when a problem comes up is
significantly larger ;-)

> 1.  *Do* they change ucol_getVersion() values in minor releases?  I
> tried to find a written policy on that.

It seems like a valid concern. The mere existence of a collation
version separate from the library major version seems to suggest that
it's possible. Perhaps they avoid it in most cases; but absent a
specific policy against it, the separate collation version seems to
allow them the freedom to do so.

> This speculation feels pretty useless.  Maybe we should go and read
> the code or ask an ICU expert, but I'm not against making it
> theoretically possible to access two different minor versions at
> once,
> just to cover all the bases for future-proofing.

I don't think this should be an overriding concern that drives the
whole design. It is a nudge in favor of search-by-collversion.

> 2.  Would package managers ever allow two minor versions to be
> installed at once?  I highly doubt it; 

Agreed.

I'm sure this has been discussed, but which distros even support
multiple major versions of ICU?

> 
> 1.  search-by-collversion:  We introduce no new "library version"
> concept to COLLATION and DATABASE object and little or no new syntax.
> Whenever opening a collation or database, the system will search some
> candidate list of ICU libraries to try to find the one that agrees
> with [dat]collversion.

[...]

> The reason I prefer major[.minor] strings over whole library names is
> that we need to dlopen two of them so it's a little easier to build
> them from those parts than have to supply both names.

It also makes it easier to know which version suffixes to look for.

>   The reason I
> prefer to keep allowing major-only versions to be listed is that it's
> good to have the option to just follow minor upgrades automatically.

Makes sense.

> Or I guess you could make something that can automatically search a
> whole directory (which directory?) to find all the suitably named
> libraries so you don't ever have to mention versions manually (if you
> want "apt-get install libicu72" to be enough with no GUC change
> needed) -- is that too weird?

That seems to go a little too far.

>   SELECT * FROM pg_available_icu_libraries()
>   SELECT * FROM pg_available_icu_collation_versions('en')

+1

> 2.  lib-version-in-providers: We introduce a separate provider value
> for each ICU version, for example ICU63, plus an unversioned ICU like
> today.

I expressed interest in this approach before, but when you allowed ICU
compiled with --disable-renaming, that mitigated my concerns about when
to throw that error.

> 3.  lib-version-in-attributes: We introduce daticuversion (alongside
> datcollversion) and collicuversion (alongside collversion).

I think this is the best among 2-4.

> 4.  lib-version-in-locale:  "63:en" from earlier versions.  That was
> mostly a strawman proposal to avoid getting bogged down in
> syntax/catalogue/model change discussions while trying to prove that
> dlopen would even work.  It doesn't sound like anyone really likes
> this.

I don't see any advantage of this over 3.

> 5.  lib-version-in-collversion:  We didn't explicitly discuss this
> before, but you hinted at it: we could just use u_getVersion() in
> [dat]collversion.

The advantage here is that it's very easy to tell the admin what
library the collation is looking for, but the disadvantages you point
out seem a lot worse: migration problems from v15, and the minor
version question.



I'd vote for 1 on the grounds that it's easier to document and
understand a single collation version, which comes straight from
ucol_getVersion(). This approach makes it a separate problem to find
the collation version among whatever libraries the admin can provide;
but adding some observability into the search should mitigate any
confusion.

Can you go over the advantages of approaches 2-4 again? Is it just a
concern about burdening the admin with finding the right ICU library
version for a given collation version? That's a valid concern, but I
don't think that should be an overriding design point. It seems more
important to model the collation versions properly.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-24 02:57  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  1 sibling, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-24 02:57 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Thu, Nov 24, 2022 at 3:07 PM Jeff Davis <[email protected]> wrote:
> I'm sure this has been discussed, but which distros even support
> multiple major versions of ICU?

For Debian and friends, you can install any number of libicuNN
packages (if you can find them eg from previous release repos), but
there's only one libicu-dev.  That means that one specific major
version is blessed by each Debian release and has its headers and
static libraries for you to use as a developer, but you can still
install the dynamic libraries from older releases at the same time to
satisfy the dependencies of packages or programs that were built on an
earlier OS release.  They don't declare conflicts on each other and
they contain non-conflicting filenames.  That's similar to the way
standard libraries and various other things are treated, for backward
compatibility.

For RHEL and friends, I'm pretty sure it's the same concept, but I
don't use those and haven't seen it with my own eyes.

I don't know for other Linux distros/families, but I expect the above
two cover a huge percentage of our users and I expect others to have
made similar choices.

For the BSDs, which tend to have a single binary package with both
headers and libraries owing to their origins as source-based
distributions (ports), the above way of thinking doesn't work; I
couldn't develop this on my usual FreeBSD battlestation without
building ICU myself (problem being that there's only one "pkg install
icu") and I hope to talk to someone who knows what to do about that
eventually.  I want this to work there easily for end users.

macOS and Windows have so many different ways of installing things
that there isn't a single answer there; supposedly open source is like
a bazaar and closed source like a cathedral, but as far as package
management goes, it looks more like rubble to me.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-24 04:48  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-24 04:48 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Thu, Nov 24, 2022 at 3:07 PM Jeff Davis <[email protected]> wrote:
> I'd vote for 1 on the grounds that it's easier to document and
> understand a single collation version, which comes straight from
> ucol_getVersion(). This approach makes it a separate problem to find
> the collation version among whatever libraries the admin can provide;
> but adding some observability into the search should mitigate any
> confusion.

OK, it sounds like I should code that up next.

> Can you go over the advantages of approaches 2-4 again? Is it just a
> concern about burdening the admin with finding the right ICU library
> version for a given collation version? That's a valid concern, but I
> don't think that should be an overriding design point. It seems more
> important to model the collation versions properly.

Yes, that's a good summary.  The user has a problem, and the solution
is to find some version of ICU and install it, so the problem space
necessarily involves the other kind of version.  My idea was that we
should therefore make that part of the model.  But the observability
support does indeed make it a bit clearer what's going on.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-26 05:27  Thomas Munro <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 3 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-26 05:27 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Thu, Nov 24, 2022 at 5:48 PM Thomas Munro <[email protected]> wrote:
> On Thu, Nov 24, 2022 at 3:07 PM Jeff Davis <[email protected]> wrote:
> > I'd vote for 1 on the grounds that it's easier to document and
> > understand a single collation version, which comes straight from
> > ucol_getVersion(). This approach makes it a separate problem to find
> > the collation version among whatever libraries the admin can provide;
> > but adding some observability into the search should mitigate any
> > confusion.
>
> OK, it sounds like I should code that up next.

Here's the first iteration.  The version rosetta stone functions look like this:

postgres=# select * from pg_icu_library_versions();
 icu_version | unicode_version | cldr_version
-------------+-----------------+--------------
 67.1        | 13.0            | 37.0
 63.1        | 11.0            | 34.0
 57.1        | 8.0             | 29.0
(3 rows)

postgres=# select * from pg_icu_collation_versions('zh');
 icu_version | uca_version | collator_version
-------------+-------------+------------------
 67.1        | 13.0        | 153.14.37
 63.1        | 11.0        | 153.88.34
 57.1        | 8.0         | 153.64.29
(3 rows)

It's no longer necessary to put anything in PG_TEST_EXTRA to run
"meson test irc/020_multiversion" usefully.  It will find extra ICU
versions all by itself in your system library search path and SKIP if
it doesn't find a second major version.  I have tried to cover the
main scenarios that I expect users to encounter in the update TAP
tests, with commentary that I hope will be helpful to assess the
usability of this thing.

Other changes:

* now using RTLD_LOCAL instead of RTLD_GLOBAL (I guess the latter
might cause trouble for someone using --disable-renaming, but I
haven't tested that and am not an expert on linker/loader arcana)
* fixed library names on Windows (based on reading the manual, but I
haven't tested that)
* fixed failure on non-ICU builds (the reason CI was failing in v7,
some misplaced #ifdefs)
* various cleanup
* I've attached a throwaway patch to install a second ICU version on
Debian/amd64 on CI, since otherwise the new test would SKIP on all
systems

This is just a first cut, but enough to try out and see if we like it,
what needs to be improved, what edge cases we haven't thought about
etc.  Let me know what you think.


Attachments:

  [text/x-patch] v8-0001-WIP-Multi-version-ICU.patch (66.3K, ../../CA+hUKGLr=9d+-k8PVv8e__TOxuq=n0SKNDpqCzbGrK5EbDyxAg@mail.gmail.com/2-v8-0001-WIP-Multi-version-ICU.patch)
  download | inline diff:
From 0d96bfbec02245ddce6c985250ff0f8d38e41df9 Mon Sep 17 00:00:00 2001
From: Thomas Munro <[email protected]>
Date: Wed, 8 Jun 2022 17:43:53 +1200
Subject: [PATCH v8 1/2] WIP: Multi-version ICU.

Add a layer of indirection when accessing ICU, so that multiple major
versions of the library can be used at once.  Versions other than the
one that PostgreSQL was linked against are opened with dlopen(), but we
refuse to open version higher than the one were were compiled against.
The ABI might change in future releases so that wouldn't be safe.

Whenever creating a DATABASE or COLLATION object that uses ICU, we'll
use the "default" ICU library and record its ucol_getVersion() in the
catalog.  That's usually the one we're linked against but another can be
selected with the setting default_icu_version_library.

Whenever opening an existing DATABASE or COLLATION object that uses ICU,
we'll see the recorded [dat]collversion and try to find the ICU library
that provides that version.  If we can't, we'll fall back to using the
default ICU library with a warning that the user should either install
another ICU library version, or rebuild affected database objects and
REFRESH.

New GUCs:

icu_library_path

  A place to find ICU libraries, if not the default system library
  search path.

icu_library_versions

  A comma-separated list of ICU major or major.minor versions to make
  available to PostgreSQL, or * for every major version that can be
  found (the default).

default_icu_library_version

  The major or major.minor version to use for new objects and as a
  fallback (with warnings) if the right version can't be found.

Reviewed-by: Peter Eisentraut <[email protected]>
Reviewed-by: Jeff Davis <[email protected]>
Discussion: https://postgr.es/m/CA%2BhUKGL4VZRpP3CkjYQkv4RQ6pRYkPkSNgKSxFBwciECQ0mEuQ%40mail.gmail.com
---
 src/backend/access/hash/hashfunc.c            |  16 +-
 src/backend/commands/collationcmds.c          |  20 +
 src/backend/utils/adt/formatting.c            |  53 +-
 src/backend/utils/adt/pg_locale.c             | 748 +++++++++++++++++-
 src/backend/utils/adt/varchar.c               |  16 +-
 src/backend/utils/adt/varlena.c               |  56 +-
 src/backend/utils/init/postinit.c             |  34 +-
 src/backend/utils/misc/guc_tables.c           |  40 +-
 src/backend/utils/misc/postgresql.conf.sample |  10 +
 src/include/catalog/pg_proc.dat               |  23 +
 src/include/utils/pg_locale.h                 |  85 +-
 src/test/icu/meson.build                      |   1 +
 src/test/icu/t/020_multiversion.pl            | 274 +++++++
 src/tools/pgindent/typedefs.list              |   4 +
 14 files changed, 1275 insertions(+), 105 deletions(-)
 create mode 100644 src/test/icu/t/020_multiversion.pl

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index b57ed946c4..0a61538efd 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -298,11 +298,11 @@ hashtext(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -355,11 +355,11 @@ hashtextextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index 81e54e0ce6..4fb0c77f38 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -853,6 +853,26 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 					CreateComments(collid, CollationRelationId, 0,
 								   icucomment);
 			}
+
+			/* Also create an object pinned to an ICU major version. */
+			collid = CollationCreate(psprintf("%s-x-icu-%d", langtag, U_ICU_VERSION_MAJOR_NUM),
+									 nspid, GetUserId(),
+									 COLLPROVIDER_ICU, true, -1,
+									 NULL, NULL,
+									 psprintf("%d:%s", U_ICU_VERSION_MAJOR_NUM, iculocstr),
+									 get_collation_actual_version(COLLPROVIDER_ICU, iculocstr),
+									 true, true);
+			if (OidIsValid(collid))
+			{
+				ncreated++;
+
+				CommandCounterIncrement();
+
+				icucomment = get_icu_locale_comment(name);
+				if (icucomment)
+					CreateComments(collid, CollationRelationId, 0,
+								   icucomment);
+			}
 		}
 	}
 #endif							/* USE_ICU */
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index 26f498b5df..0c3c7724d7 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -1599,6 +1599,11 @@ typedef int32_t (*ICU_Convert_Func) (UChar *dest, int32_t destCapacity,
 									 const UChar *src, int32_t srcLength,
 									 const char *locale,
 									 UErrorCode *pErrorCode);
+typedef int32_t (*ICU_Convert_BI_Func) (UChar *dest, int32_t destCapacity,
+										const UChar *src, int32_t srcLength,
+										UBreakIterator *bi,
+										const char *locale,
+										UErrorCode *pErrorCode);
 
 static int32_t
 icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
@@ -1623,18 +1628,41 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("case conversion failed: %s", u_errorName(status))));
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
 	return len_dest;
 }
 
+/*
+ * Like icu_convert_case, but func takes a break iterator (which we don't
+ * make use of).
+ */
 static int32_t
-u_strToTitle_default_BI(UChar *dest, int32_t destCapacity,
-						const UChar *src, int32_t srcLength,
-						const char *locale,
-						UErrorCode *pErrorCode)
+icu_convert_case_bi(ICU_Convert_BI_Func func, pg_locale_t mylocale,
+					UChar **buff_dest, UChar *buff_source, int32_t len_source)
 {
-	return u_strToTitle(dest, destCapacity, src, srcLength,
-						NULL, locale, pErrorCode);
+	UErrorCode	status;
+	int32_t		len_dest;
+
+	len_dest = len_source;		/* try first with same length */
+	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+	status = U_ZERO_ERROR;
+	len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+					mylocale->info.icu.locale, &status);
+	if (status == U_BUFFER_OVERFLOW_ERROR)
+	{
+		/* try again with adjusted length */
+		pfree(*buff_dest);
+		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+		status = U_ZERO_ERROR;
+		len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+						mylocale->info.icu.locale, &status);
+	}
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("case conversion failed: %s",
+						PG_ICU_LIB(mylocale)->errorName(status))));
+	return len_dest;
 }
 
 #endif							/* USE_ICU */
@@ -1702,7 +1730,8 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToLower, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToLower,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1824,7 +1853,8 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToUpper, mylocale,
+			len_conv = icu_convert_case(PG_ICU_LIB(mylocale)->strToUpper,
+										mylocale,
 										&buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
@@ -1947,8 +1977,9 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 			UChar	   *buff_conv;
 
 			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToTitle_default_BI, mylocale,
-										&buff_conv, buff_uchar, len_uchar);
+			len_conv = icu_convert_case_bi(PG_ICU_LIB(mylocale)->strToTitle,
+										   mylocale,
+										   &buff_conv, buff_uchar, len_uchar);
 			icu_from_uchar(&result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 2b42d9ccd8..004100af66 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -57,7 +57,9 @@
 #include "access/htup_details.h"
 #include "catalog/pg_collation.h"
 #include "catalog/pg_control.h"
+#include "funcapi.h"
 #include "mb/pg_wchar.h"
+#include "miscadmin.h"
 #include "utils/builtins.h"
 #include "utils/formatting.h"
 #include "utils/guc_hooks.h"
@@ -69,6 +71,8 @@
 
 #ifdef USE_ICU
 #include <unicode/ucnv.h>
+#include <unicode/ulocdata.h>
+#include <unicode/ustring.h>
 #endif
 
 #ifdef __GLIBC__
@@ -79,14 +83,36 @@
 #include <shlwapi.h>
 #endif
 
+#include <dlfcn.h>
+
 #define		MAX_L10N_DATA		80
 
+#ifdef USE_ICU
+
+/*
+ * We don't want to call into dlopen'd ICU libraries that are newer than the
+ * one we were compiled and linked against, just in case there is an
+ * incompatible API change.
+ */
+#define PG_MAX_ICU_MAJOR_VERSION U_ICU_VERSION_MAJOR_NUM
+
+/*
+ * The oldest ICU release we're likely to encounter, and that has all the
+ * funcitons required.
+ */
+#define PG_MIN_ICU_MAJOR_VERSION 50
+
+#endif
+
 
 /* GUC settings */
 char	   *locale_messages;
 char	   *locale_monetary;
 char	   *locale_numeric;
 char	   *locale_time;
+char	   *icu_library_path;
+char	   *icu_library_versions;
+char	   *default_icu_library_version;
 
 /*
  * lc_time localization cache.
@@ -123,7 +149,9 @@ static char *IsoLocaleName(const char *);
 #endif
 
 #ifdef USE_ICU
-static void icu_set_collation_attributes(UCollator *collator, const char *loc);
+static void icu_set_collation_attributes(pg_icu_library *lib,
+										 UCollator *collator,
+										 const char *loc);
 #endif
 
 /*
@@ -1400,33 +1428,544 @@ lc_ctype_is_c(Oid collation)
 
 struct pg_locale_struct default_locale;
 
-void
+#ifdef USE_ICU
+
+static pg_icu_library *icu_library_list;
+static pg_icu_library *default_icu_library;
+static bool icu_library_list_fully_loaded;
+
+static void *
+get_icu_function(void *handle, const char *function, int version)
+{
+	char		function_with_version[80];
+	void	   *result;
+
+	/*
+	 * Try to look it up using the symbols with major versions, but if that
+	 * doesn't work, also try the unversioned name in case the library was
+	 * configured with --disable-renaming.
+	 */
+	snprintf(function_with_version, sizeof(function_with_version), "%s_%d",
+			 function, version);
+	result = dlsym(handle, function_with_version);
+
+	return result ? result : dlsym(handle, function);
+}
+
+static void
+make_icu_library_name(char *output,
+					  const char *name,
+					  int major_version,
+					  int minor_version)
+{
+	/*
+	 * See
+	 * https://unicode-org.github.io/icu/userguide/icu4c/packaging.html#icu-versions
+	 * for conventions on library naming on POSIX and Windows systems.  Apple
+	 * isn't mentioned but varies in the usual way.
+	 *
+	 * Format 1 is expected to be a major version-only symlink pointing to a
+	 * specific minor version (or on Windows it may be the actual library).
+	 * Format 2 is expected to be an actual library.
+	 */
+#ifdef WIN32
+#define ICU_LIBRARY_NAME_FORMAT1 "%s%sicu%s%d" DLSUFFIX
+#define ICU_LIBRARY_NAME_FORMAT2 "%s%sicu%s%d.%d" DLSUFFIX
+#elif defined(__darwin__)
+#define ICU_LIBRARY_NAME_FORMAT1 "%s%slibicu%s.%d" DLSUFFIX
+#define ICU_LIBRARY_NAME_FORMAT2 "%s%slibicu%s.%d.%d" DLSUFFIX
+#else
+#define ICU_LIBRARY_NAME_FORMAT1 "%s%slibicu%s" DLSUFFIX ".%d"
+#define ICU_LIBRARY_NAME_FORMAT2 "%s%slibicu%s" DLSUFFIX ".%d.%d"
+#endif
+
+#ifdef WIN32
+#define PATH_SEPARATOR "\\"
+#define ICU_I18N "in"
+#define ICU_UC "uc"
+#else
+#define PATH_SEPARATOR "/"
+#define ICU_I18N "i18n"
+#define ICU_UC "uc"
+#endif
+
+	if (minor_version < 0)
+		snprintf(output,
+				 MAXPGPATH,
+				 ICU_LIBRARY_NAME_FORMAT1,
+				 icu_library_path,
+				 icu_library_path[0] ? PATH_SEPARATOR : "",
+				 name,
+				 major_version);
+	else
+		snprintf(output,
+				 MAXPGPATH,
+				 ICU_LIBRARY_NAME_FORMAT2,
+				 icu_library_path,
+				 icu_library_path[0] ? PATH_SEPARATOR : "",
+				 name,
+				 major_version,
+				 minor_version);
+}
+
+/*
+ * Given an ICU library major version and optionally minor version (or -1 for
+ * any), return the object we need to access all the symbols in the pair of
+ * libraries we need.  Returns NULL if the library can't be found.  Returns
+ * NULL and logs a warning if the library can be found but cannot be used for
+ * some reason.
+ */
+static pg_icu_library *
+load_icu_library(int major_version, int minor_version)
+{
+	UVersionInfo version_info;
+	pg_icu_library *lib;
+	void	   *libicui18n_handle = NULL;
+	void	   *libicuuc_handle = NULL;
+
+	/*
+	 * We don't dare open libraries outside the range that we know has an API
+	 * compatible with the headers we are compiling against.
+	 */
+	if (major_version < PG_MIN_ICU_MAJOR_VERSION ||
+		major_version > PG_MAX_ICU_MAJOR_VERSION)
+	{
+		elog(WARNING,
+			 "ICU version must be between %d and %d",
+			 PG_MIN_ICU_MAJOR_VERSION,
+			 PG_MAX_ICU_MAJOR_VERSION);
+		return NULL;
+	}
+
+	/*
+	 * We were compiled against a certain version of ICU, though the minor
+	 * version might have changed if the library was upgraded.  Does it
+	 * satisfy the request?
+	 */
+	u_getVersion(version_info);
+	if (version_info[0] == major_version &&
+		(minor_version == -1 || version_info[1] == minor_version))
+	{
+		/*
+		 * These assignments will fail to compile if an incompatible API
+		 * change is made to some future version of ICU, at which point we
+		 * might need to consider special treatment for different major
+		 * version ranges, with intermediate trampoline functions.
+		 */
+		lib = MemoryContextAllocZero(TopMemoryContext, sizeof(*lib));
+		lib->getICUVersion = u_getVersion;
+		lib->getUnicodeVersion = u_getUnicodeVersion;
+		lib->getCLDRVersion = ulocdata_getCLDRVersion;
+		lib->open = ucol_open;
+		lib->close = ucol_close;
+		lib->getCollatorVersion = ucol_getVersion;
+		lib->getUCAVersion = ucol_getUCAVersion;
+		lib->versionToString = u_versionToString;
+		lib->strcoll = ucol_strcoll;
+		lib->strcollUTF8 = ucol_strcollUTF8;
+		lib->getSortKey = ucol_getSortKey;
+		lib->nextSortKeyPart = ucol_nextSortKeyPart;
+		lib->setUTF8 = uiter_setUTF8;
+		lib->errorName = u_errorName;
+		lib->strToUpper = u_strToUpper;
+		lib->strToLower = u_strToLower;
+		lib->strToTitle = u_strToTitle;
+		lib->setAttribute = ucol_setAttribute;
+
+		/*
+		 * Also assert the size of a couple of types used as output buffers,
+		 * as a canary to tell us to add extra padding in the (unlikely) event
+		 * that a later release makes these values smaller.
+		 */
+		StaticAssertStmt(U_MAX_VERSION_STRING_LENGTH == 20,
+						 "u_versionToString output buffer size changed incompatibly");
+		StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
+						 "ucol_getVersion output buffer size changed incompatibly");
+	}
+	else
+	{
+		/* This is an older version, so we'll need to use dlopen(). */
+		char		libicui18n_name[MAXPGPATH];
+		char		libicuuc_name[MAXPGPATH];
+
+		/* Load the internationalization library. */
+		make_icu_library_name(libicui18n_name, ICU_I18N, major_version, minor_version);
+		libicui18n_handle = dlopen(libicui18n_name, RTLD_NOW | RTLD_LOCAL);
+		if (!libicui18n_handle)
+			return NULL;
+
+		/* Load the common library. */
+		make_icu_library_name(libicuuc_name, ICU_UC, major_version, minor_version);
+		libicuuc_handle = dlopen(libicuuc_name, RTLD_NOW | RTLD_LOCAL);
+		if (!libicui18n_handle)
+		{
+			elog(WARNING, "found library \"%s\" but not companion library \"%s\"",
+				 libicui18n_name, libicuuc_name);
+			dlclose(libicui18n_handle);
+			return NULL;
+		}
+
+		/*
+		 * We only allocate the pg_icu_library object after successfully
+		 * opening the libraries to minimize the work done in the ENOENT case,
+		 * when probing a range of versions.  That means we might need to
+		 * clean up on allocation failure.
+		 */
+		lib = MemoryContextAllocExtended(TopMemoryContext, sizeof(*lib),
+										 MCXT_ALLOC_NO_OOM);
+		if (!lib)
+		{
+			dlclose(libicui18n_handle);
+			dlclose(libicuuc_handle);
+			elog(ERROR, "out of memory");
+		}
+
+		/* Now try to find all the symbols we need. */
+		lib->getICUVersion = get_icu_function(libicui18n_handle,
+											  "u_getVersion",
+											  major_version);
+		lib->getUnicodeVersion = get_icu_function(libicui18n_handle,
+												  "u_getUnicodeVersion",
+												  major_version);
+		lib->getCLDRVersion = get_icu_function(libicui18n_handle,
+											   "ulocdata_getCLDRVersion",
+											   major_version);
+		lib->open = get_icu_function(libicui18n_handle,
+									 "ucol_open",
+									 major_version);
+		lib->close = get_icu_function(libicui18n_handle,
+									  "ucol_close",
+									  major_version);
+		lib->getCollatorVersion = get_icu_function(libicui18n_handle,
+												   "ucol_getVersion",
+												   major_version);
+		lib->getUCAVersion = get_icu_function(libicui18n_handle,
+											  "ucol_getUCAVersion",
+											  major_version);
+		lib->versionToString = get_icu_function(libicui18n_handle,
+												"u_versionToString",
+												major_version);
+		lib->strcoll = get_icu_function(libicui18n_handle,
+										"ucol_strcoll",
+										major_version);
+		lib->strcollUTF8 = get_icu_function(libicui18n_handle,
+											"ucol_strcollUTF8",
+											major_version);
+		lib->getSortKey = get_icu_function(libicui18n_handle,
+										   "ucol_getSortKey",
+										   major_version);
+		lib->nextSortKeyPart = get_icu_function(libicui18n_handle,
+												"ucol_nextSortKeyPart",
+												major_version);
+		lib->setUTF8 = get_icu_function(libicui18n_handle,
+										"uiter_setUTF8",
+										major_version);
+		lib->errorName = get_icu_function(libicui18n_handle,
+										  "u_errorName",
+										  major_version);
+		lib->strToUpper = get_icu_function(libicuuc_handle,
+										   "u_strToUpper",
+										   major_version);
+		lib->strToLower = get_icu_function(libicuuc_handle,
+										   "u_strToLower",
+										   major_version);
+		lib->strToTitle = get_icu_function(libicuuc_handle,
+										   "u_strToTitle",
+										   major_version);
+		lib->setAttribute = get_icu_function(libicui18n_handle,
+											 "ucol_setAttribute",
+											 major_version);
+
+		/* Did we find everything? */
+		if (!lib->getICUVersion ||
+			!lib->getUnicodeVersion ||
+			!lib->getCLDRVersion ||
+			!lib->open ||
+			!lib->close ||
+			!lib->getCollatorVersion ||
+			!lib->getUCAVersion ||
+			!lib->versionToString ||
+			!lib->strcoll ||
+			!lib->strcollUTF8 ||
+			!lib->getSortKey ||
+			!lib->nextSortKeyPart ||
+			!lib->setUTF8 ||
+			!lib->errorName ||
+			!lib->strToUpper ||
+			!lib->strToLower ||
+			!lib->strToTitle ||
+			!lib->setAttribute)
+		{
+			dlclose(libicui18n_handle);
+			dlclose(libicuuc_handle);
+			pfree(lib);
+			ereport(WARNING,
+					(errmsg("could not find all expected symbols in libraries \"%s\" and \"%s\"",
+							libicui18n_name, libicuuc_name)));
+			return NULL;
+		}
+	}
+
+	/* Is this major.minor already loaded? */
+	lib->getICUVersion(version_info);
+	lib->major_version = version_info[0];
+	lib->minor_version = version_info[1];
+	for (pg_icu_library *lib2 = icu_library_list; lib2; lib2 = lib2->next)
+	{
+		if (lib2->major_version == lib->major_version &&
+			lib2->minor_version == lib->minor_version)
+		{
+			if (libicui18n_handle)
+				dlclose(libicui18n_handle);
+			if (libicuuc_handle)
+				dlclose(libicuuc_handle);
+			pfree(lib);
+
+			/* Return the one we already had. */
+			return lib2;
+		}
+	}
+
+	/* Add to list of loaded libraries. */
+	lib->next = icu_library_list;
+	icu_library_list = lib;
+
+	return lib;
+}
+
+static pg_icu_library *
+get_icu_library_list(void)
+{
+	char	   *copy;
+	char	   *token;
+	char	   *saveptr;
+
+	if (icu_library_list_fully_loaded)
+		return icu_library_list;
+
+	copy = pstrdup(icu_library_versions);
+	token = strtok_r(copy, ",", &saveptr);
+	while (token)
+	{
+		int			major_version;
+		int			minor_version;
+
+		/* Ignore spaces between commas. */
+		while (*token == ' ')
+			++token;
+
+		if (strcmp(token, "*") == 0)
+		{
+			/* Try to load every supportable major library version. */
+			for (int i = PG_MIN_ICU_MAJOR_VERSION; i <= PG_MAX_ICU_MAJOR_VERSION; ++i)
+				load_icu_library(i, -1);
+		}
+		else if (sscanf(token, "%d.%d", &major_version, &minor_version) == 2)
+		{
+			/* Try to load a version with an explicit minor version provided. */
+			if (!load_icu_library(major_version, minor_version))
+				ereport(WARNING,
+						(errmsg("could not open ICU library \"%s\"", token)));
+		}
+		else if (sscanf(token, "%d", &major_version) == 1)
+		{
+			/* Try to load a major version through symlinks. */
+			if (!load_icu_library(major_version, -1))
+				ereport(WARNING,
+						(errmsg("could not open ICU library \"%s\"", token)));
+		}
+		else
+			ereport(WARNING,
+					(errmsg("could not parse ICU library version \"%s\"", token)));
+
+		token = strtok_r(NULL, ",", &saveptr);
+	}
+	pfree(copy);
+
+	icu_library_list_fully_loaded = true;
+
+	return icu_library_list;
+}
+
+static pg_icu_library *
+get_default_icu_library(void)
+{
+	int			major_version;
+	int			minor_version;
+
+	if (default_icu_library)
+		return default_icu_library;
+
+	if (default_icu_library_version[0] == 0)
+	{
+		/* Use the linked version by default. */
+		default_icu_library = load_icu_library(PG_MAX_ICU_MAJOR_VERSION, -1);
+		Assert(default_icu_library);
+	}
+	else if (sscanf(default_icu_library_version, "%d.%d", &major_version, &minor_version) == 2)
+	{
+		/* Try to load a version with an explicit major.minor version. */
+		default_icu_library = load_icu_library(major_version, minor_version);
+	}
+	else if (sscanf(default_icu_library_version, "%d", &major_version) == 1)
+	{
+		/* Try to load a version using only major (usually a symlink on Unix). */
+		default_icu_library = load_icu_library(major_version, -1);
+	}
+	else
+	{
+		ereport(WARNING,
+				(errmsg("could not parse default_icu_library_version \"%s\"",
+						default_icu_library_version)));
+	}
+
+	if (!default_icu_library_version)
+	{
+		/*
+		 * Fall back to the linked version with a warning if the above
+		 * attempts failed.
+		 */
+		default_icu_library = load_icu_library(PG_MAX_ICU_MAJOR_VERSION, -1);
+		Assert(default_icu_library);
+		ereport(WARNING,
+				(errmsg("could not load ICU library version \"%s\", so using linked version %d.%d instead",
+						default_icu_library_version,
+						default_icu_library->major_version,
+						default_icu_library->minor_version)));
+		Assert(default_icu_library);
+	}
+
+	return default_icu_library;
+}
+
+/*
+ * Try to open a collator with a specific version from a given library.
+ * Returns NULL on failure.
+ */
+static UCollator *
+get_icu_collator(pg_icu_library *lib,
+				 const char *locale,
+				 const char *collversion)
+{
+	UErrorCode	status;
+	UCollator  *collator;
+	UVersionInfo version_info;
+	char		version_info_string[U_MAX_VERSION_STRING_LENGTH];
+
+	/* Can we even open it? */
+	status = U_ZERO_ERROR;
+	collator = lib->open(locale, &status);
+	if (!collator)
+		return NULL;
+
+	/*
+	 * Does it have the requested version?  We tolerate a null collversion
+	 * argument only for bootrapping in initdb --locale-provider=icu, where we
+	 * accept the first library we try.
+	 */
+	if (collversion)
+	{
+		lib->getCollatorVersion(collator, version_info);
+		lib->versionToString(version_info, version_info_string);
+		if (strcmp(version_info_string, collversion) != 0)
+		{
+			lib->close(collator);
+			return NULL;
+		}
+	} else
+		Assert(!IsUnderPostmaster);
+
+	/* XXX this can raise an error and leak collator! */
+	if (lib->major_version < 54)
+		icu_set_collation_attributes(lib, collator, locale);
+
+	return collator;
+}
+
+#endif
+
+/*
+ * Returns true if a collator with u_getVersion() matching collversion could
+ * not be found in any available ICU library, so the default library was used
+ * instead.
+ */
+bool
 make_icu_collator(const char *iculocstr,
+				  const char *collversion,
 				  struct pg_locale_struct *resultp)
 {
+	bool		using_default = false;
 #ifdef USE_ICU
-	UCollator  *collator;
-	UErrorCode	status;
+	pg_icu_library *lib = NULL;
+	UCollator  *collator = NULL;
 
-	status = U_ZERO_ERROR;
-	collator = ucol_open(iculocstr, &status);
-	if (U_FAILURE(status))
-		ereport(ERROR,
-				(errmsg("could not open collator for locale \"%s\": %s",
-						iculocstr, u_errorName(status))));
+	/*
+	 * Try the default library first, which might avoid the need to dlopen()
+	 * libraries in the common case that it's the version we're linked
+	 * against.
+	 */
+	lib = get_default_icu_library();
+	collator = get_icu_collator(lib, iculocstr, collversion);
 
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
-		icu_set_collation_attributes(collator, iculocstr);
+	/*
+	 * If that didn't succeed, try every available library.
+	 */
+	if (!collator)
+	{
+		for (lib = get_icu_library_list(); lib; lib = lib->next)
+		{
+			collator = get_icu_collator(lib, iculocstr, collversion);
+			if (collator)
+				break;
+		}
+	}
+
+	/*
+	 * If we didn't find a match, it's time to fall back to our default
+	 * library.  We'll also return true so the caller can generate a more
+	 * specific warning about what to do.
+	 */
+	if (!collator)
+	{
+		UVersionInfo version_info;
+		char		version_info_string[U_MAX_VERSION_STRING_LENGTH];
+		UErrorCode	status;
+
+		lib = get_default_icu_library();
+
+		status = U_ZERO_ERROR;
+		collator = lib->open(iculocstr, &status);
+		if (!collator)
+			ereport(ERROR,
+					(errmsg("could not open collator for locale \"%s\": %s",
+							iculocstr, lib->errorName(status))));
+
+		lib->getCollatorVersion(collator, version_info);
+		lib->versionToString(version_info, version_info_string);
+		ereport(WARNING,
+				(errmsg("could not find ICU collator for locale \"%s\" with "
+						"version %s, so using version %s from default "
+						"ICU library %d.%d instead",
+						iculocstr, collversion, version_info_string,
+						lib->major_version, lib->minor_version)));
+
+		using_default = true;
+	}
+
+	Assert(lib);
 
 	/* We will leak this string if the caller errors later :-( */
 	resultp->info.icu.locale = MemoryContextStrdup(TopMemoryContext, iculocstr);
 	resultp->info.icu.ucol = collator;
+	resultp->info.icu.lib = lib;
 #else							/* not USE_ICU */
 	/* could get here if a collation was created by a build with ICU */
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
 			 errmsg("ICU is not supported in this build")));
 #endif							/* not USE_ICU */
+
+	return using_default;
 }
 
 
@@ -1504,6 +2043,7 @@ pg_newlocale_from_collation(Oid collid)
 		pg_locale_t resultp;
 		Datum		datum;
 		bool		isnull;
+		const char *collversion;
 
 		tp = SearchSysCache1(COLLOID, ObjectIdGetDatum(collid));
 		if (!HeapTupleIsValid(tp))
@@ -1515,6 +2055,10 @@ pg_newlocale_from_collation(Oid collid)
 		result.provider = collform->collprovider;
 		result.deterministic = collform->collisdeterministic;
 
+		datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collversion,
+								&isnull);
+		collversion = isnull ? NULL : TextDatumGetCString(datum);
+
 		if (collform->collprovider == COLLPROVIDER_LIBC)
 		{
 #ifdef HAVE_LOCALE_T
@@ -1584,23 +2128,37 @@ pg_newlocale_from_collation(Oid collid)
 			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_colliculocale, &isnull);
 			Assert(!isnull);
 			iculocstr = TextDatumGetCString(datum);
-			make_icu_collator(iculocstr, &result);
+			if (!collversion)
+				elog(ERROR, "ICU collation lacks version");
+			if (make_icu_collator(iculocstr, collversion, &result))
+			{
+				ereport(WARNING,
+						errmsg("collation \"%s\" version mismatch",
+							   NameStr(collform->collname)),
+						errdetail("The collation in the database was created using "
+								  "locale \"%s\" version %s, "
+								  "but no ICU library with a matching collator is available",
+								  iculocstr, collversion),
+						errhint("Install a version of ICU that provides locale \"%s\" "
+								"version %s, or rebuild all objects "
+								"affected by this collation and run "
+								"ALTER COLLATION %s REFRESH VERSION.",
+								iculocstr, collversion,
+								quote_qualified_identifier(get_namespace_name(collform->collnamespace),
+														   NameStr(collform->collname))));
+			}
 		}
 
-		datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collversion,
-								&isnull);
-		if (!isnull)
+		if (collform->collprovider == COLLPROVIDER_LIBC && collversion)
 		{
 			char	   *actual_versionstr;
-			char	   *collversionstr;
+			char	   *locale;
 
-			collversionstr = TextDatumGetCString(datum);
-
-			datum = SysCacheGetAttr(COLLOID, tp, collform->collprovider == COLLPROVIDER_ICU ? Anum_pg_collation_colliculocale : Anum_pg_collation_collcollate, &isnull);
+			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collcollate, &isnull);
 			Assert(!isnull);
+			locale = TextDatumGetCString(datum);
 
-			actual_versionstr = get_collation_actual_version(collform->collprovider,
-															 TextDatumGetCString(datum));
+			actual_versionstr = get_collation_actual_version(collform->collprovider, locale);
 			if (!actual_versionstr)
 			{
 				/*
@@ -1613,18 +2171,20 @@ pg_newlocale_from_collation(Oid collid)
 								NameStr(collform->collname))));
 			}
 
-			if (strcmp(actual_versionstr, collversionstr) != 0)
+			if (strcmp(actual_versionstr, collversion) != 0)
+			{
 				ereport(WARNING,
 						(errmsg("collation \"%s\" has version mismatch",
 								NameStr(collform->collname)),
 						 errdetail("The collation in the database was created using version %s, "
 								   "but the operating system provides version %s.",
-								   collversionstr, actual_versionstr),
+								   collversion, actual_versionstr),
 						 errhint("Rebuild all objects affected by this collation and run "
 								 "ALTER COLLATION %s REFRESH VERSION, "
 								 "or build PostgreSQL with the right library version.",
 								 quote_qualified_identifier(get_namespace_name(collform->collnamespace),
 															NameStr(collform->collname)))));
+			}
 		}
 
 		ReleaseSysCache(tp);
@@ -1651,21 +2211,27 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 #ifdef USE_ICU
 	if (collprovider == COLLPROVIDER_ICU)
 	{
+		pg_icu_library *lib;
 		UCollator  *collator;
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
 
+		/*
+		 * Use the default library, but other versions might also be active
+		 * and can be seen with pg_icu_collation_versions().
+		 */
+		lib = get_default_icu_library();
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collcollate, &status);
+		collator = lib->open(collcollate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collcollate, u_errorName(status))));
-		ucol_getVersion(collator, versioninfo);
-		ucol_close(collator);
+							collcollate, lib->errorName(status))));
+		lib->getCollatorVersion(collator, versioninfo);
+		lib->close(collator);
 
-		u_versionToString(versioninfo, buf);
+		lib->versionToString(versioninfo, buf);
 		collversion = pstrdup(buf);
 	}
 	else
@@ -1731,8 +2297,110 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 	return collversion;
 }
 
+Datum
+pg_icu_library_versions(PG_FUNCTION_ARGS)
+{
+#ifdef USE_ICU
+#define PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS 3
+	ReturnSetInfo *rsinfo = (ReturnSetInfo *) fcinfo->resultinfo;
+	Datum		values[PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS];
+	bool		nulls[PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS];
+
+	InitMaterializedSRF(fcinfo, 0);
+
+	for (pg_icu_library *lib = get_icu_library_list(); lib; lib = lib->next)
+	{
+		UErrorCode	status;
+		UVersionInfo version_info;
+		char		version_string[U_MAX_VERSION_STRING_LENGTH];
+
+		lib->getICUVersion(version_info);
+		lib->versionToString(version_info, version_string);
+		values[0] = PointerGetDatum(cstring_to_text(version_string));
+		nulls[0] = false;
+
+		lib->getUnicodeVersion(version_info);
+		lib->versionToString(version_info, version_string);
+		values[1] = PointerGetDatum(cstring_to_text(version_string));
+		nulls[1] = false;
+
+		status = U_ZERO_ERROR;
+		lib->getCLDRVersion(version_info, &status);
+		if (U_SUCCESS(status))
+		{
+			lib->versionToString(version_info, version_string);
+			values[2] = PointerGetDatum(cstring_to_text(version_string));
+			nulls[2] = false;
+		}
+		else
+		{
+			nulls[2] = true;
+		}
+
+		tuplestore_putvalues(rsinfo->setResult, rsinfo->setDesc, values, nulls);
+	}
+#endif
+
+	return (Datum) 0;
+}
+
+Datum
+pg_icu_collation_versions(PG_FUNCTION_ARGS)
+{
+#ifdef USE_ICU
+#define PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS 3
+	const char *locale = text_to_cstring(PG_GETARG_TEXT_PP(0));
+	ReturnSetInfo *rsinfo = (ReturnSetInfo *) fcinfo->resultinfo;
+	Datum		values[PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS];
+	bool		nulls[PG_ICU_AVAILABLE_ICU_LIRBARIES_COLS];
+
+	InitMaterializedSRF(fcinfo, 0);
+
+	for (pg_icu_library *lib = get_icu_library_list(); lib; lib = lib->next)
+	{
+		UErrorCode	status;
+		UCollator  *collator;
+		UVersionInfo version_info;
+		char		version_string[U_MAX_VERSION_STRING_LENGTH];
+
+		status = U_ZERO_ERROR;
+		collator = lib->open(locale, &status);
+		if (!collator)
+		{
+			if (U_FAILURE(status))
+				ereport(WARNING,
+						(errmsg("could not open collator for locale \"%s\" from ICU %d.%d: %s",
+								locale,
+								lib->major_version,
+								lib->minor_version,
+								lib->errorName(status))));
+			continue;
+		}
+
+		lib->getICUVersion(version_info);
+		lib->versionToString(version_info, version_string);
+		values[0] = PointerGetDatum(cstring_to_text(version_string));
+		nulls[0] = false;
+
+		lib->getUCAVersion(collator, version_info);
+		lib->versionToString(version_info, version_string);
+		values[1] = PointerGetDatum(cstring_to_text(version_string));
+		nulls[1] = false;
+
+		lib->getCollatorVersion(collator, version_info);
+		lib->versionToString(version_info, version_string);
+		values[2] = PointerGetDatum(cstring_to_text(version_string));
+		nulls[2] = false;
+
+		tuplestore_putvalues(rsinfo->setResult, rsinfo->setDesc, values, nulls);
+	}
+#endif
+
+	return (Datum) 0;
+}
 
 #ifdef USE_ICU
+
 /*
  * Converter object for converting between ICU's UChar strings and C strings
  * in database encoding.  Since the database encoding doesn't change, we only
@@ -1855,9 +2523,10 @@ icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar)
  * ucol_open(), so this is only necessary for emulating this behavior on older
  * versions.
  */
-pg_attribute_unused()
 static void
-icu_set_collation_attributes(UCollator *collator, const char *loc)
+icu_set_collation_attributes(pg_icu_library *lib,
+							 UCollator *collator,
+							 const char *loc)
 {
 	char	   *str = asc_tolower(loc, strlen(loc));
 
@@ -1886,6 +2555,8 @@ icu_set_collation_attributes(UCollator *collator, const char *loc)
 
 			/*
 			 * See attribute name and value lists in ICU i18n/coll.cpp
+			 *
+			 * XXX Are these enumerator values stable across releases?
 			 */
 			if (strcmp(name, "colstrength") == 0)
 				uattr = UCOL_STRENGTH;
@@ -1931,7 +2602,7 @@ icu_set_collation_attributes(UCollator *collator, const char *loc)
 				status = U_ILLEGAL_ARGUMENT_ERROR;
 
 			if (status == U_ZERO_ERROR)
-				ucol_setAttribute(collator, uattr, uvalue, &status);
+				lib->setAttribute(collator, uattr, uvalue, &status);
 
 			/*
 			 * Pretend the error came from ucol_open(), for consistent error
@@ -1940,7 +2611,7 @@ icu_set_collation_attributes(UCollator *collator, const char *loc)
 			if (U_FAILURE(status))
 				ereport(ERROR,
 						(errmsg("could not open collator for locale \"%s\": %s",
-								loc, u_errorName(status))));
+								loc, lib->errorName(status))));
 		}
 	}
 }
@@ -1954,19 +2625,18 @@ void
 check_icu_locale(const char *icu_locale)
 {
 #ifdef USE_ICU
+	pg_icu_library *lib;
 	UCollator  *collator;
 	UErrorCode	status;
 
+	lib = get_default_icu_library();
 	status = U_ZERO_ERROR;
-	collator = ucol_open(icu_locale, &status);
+	collator = lib->open(icu_locale, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open collator for locale \"%s\": %s",
-						icu_locale, u_errorName(status))));
-
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
-		icu_set_collation_attributes(collator, icu_locale);
-	ucol_close(collator);
+						icu_locale, lib->errorName(status))));
+	lib->close(collator);
 #else
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index 68e2e6f7a7..e0c86870e0 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -1026,11 +1026,11 @@ hashbpchar(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, keydata, keylen);
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any(buf, bsize);
 
@@ -1087,11 +1087,11 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 
 			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
 
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+													 uchar, ulen, NULL, 0);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
+			PG_ICU_LIB(mylocale)->getSortKey(PG_ICU_COL(mylocale),
+											 uchar, ulen, buf, bsize);
 
 			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
 
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c5e7ee7ca2..cf891a5654 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1667,13 +1667,14 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					UErrorCode	status;
 
 					status = U_ZERO_ERROR;
-					result = ucol_strcollUTF8(mylocale->info.icu.ucol,
-											  arg1, len1,
-											  arg2, len2,
-											  &status);
+					result = PG_ICU_LIB(mylocale)->strcollUTF8(PG_ICU_COL(mylocale),
+															   arg1, len1,
+															   arg2, len2,
+															   &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
-								(errmsg("collation failed: %s", u_errorName(status))));
+								(errmsg("collation failed: %s",
+										PG_ICU_LIB(mylocale)->errorName(status))));
 				}
 				else
 #endif
@@ -1686,9 +1687,9 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 					ulen1 = icu_to_uchar(&uchar1, arg1, len1);
 					ulen2 = icu_to_uchar(&uchar2, arg2, len2);
 
-					result = ucol_strcoll(mylocale->info.icu.ucol,
-										  uchar1, ulen1,
-										  uchar2, ulen2);
+					result = PG_ICU_LIB(mylocale)->strcoll(PG_ICU_COL(mylocale),
+														   uchar1, ulen1,
+														   uchar2, ulen2);
 
 					pfree(uchar1);
 					pfree(uchar2);
@@ -2388,13 +2389,14 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				UErrorCode	status;
 
 				status = U_ZERO_ERROR;
-				result = ucol_strcollUTF8(sss->locale->info.icu.ucol,
-										  a1p, len1,
-										  a2p, len2,
-										  &status);
+				result = PG_ICU_LIB(sss->locale)->strcollUTF8(PG_ICU_COL(sss->locale),
+															  a1p, len1,
+															  a2p, len2,
+															  &status);
 				if (U_FAILURE(status))
 					ereport(ERROR,
-							(errmsg("collation failed: %s", u_errorName(status))));
+							(errmsg("collation failed: %s",
+									PG_ICU_LIB(sss->locale)->errorName(status))));
 			}
 			else
 #endif
@@ -2407,9 +2409,9 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 				ulen1 = icu_to_uchar(&uchar1, a1p, len1);
 				ulen2 = icu_to_uchar(&uchar2, a2p, len2);
 
-				result = ucol_strcoll(sss->locale->info.icu.ucol,
-									  uchar1, ulen1,
-									  uchar2, ulen2);
+				result = PG_ICU_LIB(sss->locale)->strcoll(PG_ICU_COL(sss->locale),
+														  uchar1, ulen1,
+														  uchar2, ulen2);
 
 				pfree(uchar1);
 				pfree(uchar2);
@@ -2569,24 +2571,24 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 					uint32_t	state[2];
 					UErrorCode	status;
 
-					uiter_setUTF8(&iter, sss->buf1, len);
+					PG_ICU_LIB(sss->locale)->setUTF8(&iter, sss->buf1, len);
 					state[0] = state[1] = 0;	/* won't need that again */
 					status = U_ZERO_ERROR;
-					bsize = ucol_nextSortKeyPart(sss->locale->info.icu.ucol,
-												 &iter,
-												 state,
-												 (uint8_t *) sss->buf2,
-												 Min(sizeof(Datum), sss->buflen2),
-												 &status);
+					bsize = PG_ICU_LIB(sss->locale)->nextSortKeyPart(PG_ICU_COL(sss->locale),
+																	 &iter,
+																	 state,
+																	 (uint8_t *) sss->buf2,
+																	 Min(sizeof(Datum), sss->buflen2),
+																	 &status);
 					if (U_FAILURE(status))
 						ereport(ERROR,
 								(errmsg("sort key generation failed: %s",
-										u_errorName(status))));
+										PG_ICU_LIB(sss->locale)->errorName(status))));
 				}
 				else
-					bsize = ucol_getSortKey(sss->locale->info.icu.ucol,
-											uchar, ulen,
-											(uint8_t *) sss->buf2, sss->buflen2);
+					bsize = PG_ICU_LIB(sss->locale)->getSortKey(PG_ICU_COL(sss->locale),
+																uchar, ulen,
+																(uint8_t *) sss->buf2, sss->buflen2);
 			}
 			else
 #endif
diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c
index a990c833c5..236ec6d682 100644
--- a/src/backend/utils/init/postinit.c
+++ b/src/backend/utils/init/postinit.c
@@ -317,6 +317,7 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 	char	   *collate;
 	char	   *ctype;
 	char	   *iculocale;
+	char	   *collversion;
 
 	/* Fetch our pg_database row normally, via syscache */
 	tup = SearchSysCache1(DATABASEOID, ObjectIdGetDatum(MyDatabaseId));
@@ -404,6 +405,9 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 	datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_datctype, &isnull);
 	Assert(!isnull);
 	ctype = TextDatumGetCString(datum);
+	datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_datcollversion,
+							&isnull);
+	collversion = isnull ? NULL : TextDatumGetCString(datum);
 
 	if (pg_perm_setlocale(LC_COLLATE, collate) == NULL)
 		ereport(FATAL,
@@ -424,7 +428,20 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 		datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_daticulocale, &isnull);
 		Assert(!isnull);
 		iculocale = TextDatumGetCString(datum);
-		make_icu_collator(iculocale, &default_locale);
+		if (make_icu_collator(iculocale, collversion, &default_locale))
+		{
+			ereport(WARNING,
+					errmsg("database \"%s\" has a collation version mismatch",
+						   name),
+					errdetail("The database was created using ICU locale \"%s\" version %s, "
+							  "but no ICU library with a matching collator is available",
+							  iculocale, collversion),
+					errhint("Install a version of ICU that provides locale \"%s\" "
+							"version %s, or rebuild all objects "
+							"in this database that use the default collation and run "
+							"ALTER DATABASE %s REFRESH COLLATION VERSION.",
+							iculocale, collversion, name));
+		}
 	}
 	else
 		iculocale = NULL;
@@ -443,32 +460,29 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 	 * pg_newlocale_from_collation().  Note that here we warn instead of error
 	 * in any case, so that we don't prevent connecting.
 	 */
-	datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_datcollversion,
-							&isnull);
-	if (!isnull)
+	if (dbform->datlocprovider == COLLPROVIDER_LIBC && collversion)
 	{
 		char	   *actual_versionstr;
-		char	   *collversionstr;
-
-		collversionstr = TextDatumGetCString(datum);
 
-		actual_versionstr = get_collation_actual_version(dbform->datlocprovider, dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : collate);
+		actual_versionstr = get_collation_actual_version(dbform->datlocprovider, collate);
 		if (!actual_versionstr)
 			/* should not happen */
 			elog(WARNING,
 				 "database \"%s\" has no actual collation version, but a version was recorded",
 				 name);
-		else if (strcmp(actual_versionstr, collversionstr) != 0)
+		else if (strcmp(actual_versionstr, collversion) != 0)
+		{
 			ereport(WARNING,
 					(errmsg("database \"%s\" has a collation version mismatch",
 							name),
 					 errdetail("The database was created using collation version %s, "
 							   "but the operating system provides version %s.",
-							   collversionstr, actual_versionstr),
+							   collversion, actual_versionstr),
 					 errhint("Rebuild all objects in this database that use the default collation and run "
 							 "ALTER DATABASE %s REFRESH COLLATION VERSION, "
 							 "or build PostgreSQL with the right library version.",
 							 quote_identifier(name))));
+		}
 	}
 
 	/* Make the locale settings visible as GUC variables, too */
diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c
index 349dd6a537..5fd08f7693 100644
--- a/src/backend/utils/misc/guc_tables.c
+++ b/src/backend/utils/misc/guc_tables.c
@@ -3941,7 +3941,45 @@ struct config_string ConfigureNamesString[] =
 	},
 
 	{
-		{"krb_server_keyfile", PGC_SIGHUP, CONN_AUTH_AUTH,
+		{"icu_library_path", PGC_SUSET, COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Sets the path for dynamically loadable ICU libraries."),
+			gettext_noop("If versions of ICU other than the one that "
+						 "PostgreSQL is linked against are needed, they will "
+						 "be opened from this directory.  If empty, the "
+						 "system linker search path will be used."),
+			GUC_SUPERUSER_ONLY
+		},
+		&icu_library_path,
+		"",
+		NULL, NULL, NULL
+	},
+
+	{
+		{"icu_library_versions", PGC_SUSET, COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Sets the available ICU library versions."),
+			gettext_noop("A comma-separated list of major or major.minor ICU versions "
+						 "that will be searched for referenced collation versions.  Use * "
+						 "for all possible versions."),
+			GUC_SUPERUSER_ONLY
+		},
+		&icu_library_versions,
+		"*",
+		NULL, NULL, NULL
+	},
+
+	{
+		{"default_icu_library_version", PGC_SIGHUP, COMPAT_OPTIONS_PREVIOUS,
+			gettext_noop("Sets the ICU library version used to create new collations and databases."),
+			gettext_noop("A major or major.minor ICU version, or empty string for the linked version."),
+			GUC_SUPERUSER_ONLY
+		},
+		&default_icu_library_version,
+		"",
+		NULL, NULL, NULL
+	},
+
+	{
+		{"krb_server_keyfile", PGC_POSTMASTER, CONN_AUTH_AUTH,
 			gettext_noop("Sets the location of the Kerberos server key file."),
 			NULL,
 			GUC_SUPERUSER_ONLY
diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample
index 868d21c351..93a0ad6406 100644
--- a/src/backend/utils/misc/postgresql.conf.sample
+++ b/src/backend/utils/misc/postgresql.conf.sample
@@ -727,6 +727,16 @@
 #lc_numeric = 'C'			# locale for number formatting
 #lc_time = 'C'				# locale for time formatting
 
+#icu_library_path = ''			# path for dynamically loaded ICU
+					# libraries
+#icu_library_versions = '*'		# comma-separated list of ICU major
+					# or major.minor versions to make
+					# available, or * for all major
+					# versions that can be found
+#default_icu_library_version = ''	# version of ICU to use for new
+					# databases and collations, defaults
+					# to the latest version
+
 # default configuration for text search
 #default_text_search_config = 'pg_catalog.simple'
 
diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat
index f9301b2627..607968b340 100644
--- a/src/include/catalog/pg_proc.dat
+++ b/src/include/catalog/pg_proc.dat
@@ -11733,6 +11733,29 @@
   proname => 'pg_database_collation_actual_version', procost => '100',
   provolatile => 'v', prorettype => 'text', proargtypes => 'oid',
   prosrc => 'pg_database_collation_actual_version' },
+{ oid => '8888', descr => 'get available ICU library versions',
+  proname => 'pg_icu_library_versions', prorettype => 'record',
+  procost => '10',
+  prorows => '2',
+  proretset => 't',
+  provolatile => 'v',
+  proargtypes => '',
+  proallargtypes => '{text,text,text}',
+  proargmodes => '{o,o,o}',
+  proargnames => '{icu_version,unicode_version,cldr_version}',
+  prosrc => 'pg_icu_library_versions' },
+{ oid => '8889', descr => 'get available ICU collation versions',
+  proname => 'pg_icu_collation_versions', prorettype => 'record',
+  procost => '10',
+  prorows => '2',
+  proretset => 't',
+  provolatile => 'v',
+  proargtypes => 'text',
+  proallargtypes => '{text,text,text,text}',
+  proargmodes => '{i,o,o,o}',
+  proargnames => '{locale,icu_version,uca_version,collator_version}',
+  prosrc => 'pg_icu_collation_versions' },
+
 
 # system management/monitoring related functions
 { oid => '3353', descr => 'list files in the log directory',
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index a875942123..554b335df9 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -17,6 +17,7 @@
 #endif
 #ifdef USE_ICU
 #include <unicode/ucol.h>
+#include <unicode/ubrk.h>
 #endif
 
 #ifdef USE_ICU
@@ -40,6 +41,9 @@ extern PGDLLIMPORT char *locale_messages;
 extern PGDLLIMPORT char *locale_monetary;
 extern PGDLLIMPORT char *locale_numeric;
 extern PGDLLIMPORT char *locale_time;
+extern PGDLLIMPORT char *icu_library_path;
+extern PGDLLIMPORT char *icu_library_versions;
+extern PGDLLIMPORT char *default_icu_library_version;
 
 /* lc_time localization cache */
 extern PGDLLIMPORT char *localized_abbrev_days[];
@@ -63,6 +67,78 @@ extern struct lconv *PGLC_localeconv(void);
 
 extern void cache_locale_time(void);
 
+#ifdef USE_ICU
+
+/*
+ * An ICU library version that we're either linked against or have loaded at
+ * runtime.
+ */
+typedef struct pg_icu_library
+{
+	int			major_version;
+	int			minor_version;
+	void		(*getICUVersion) (UVersionInfo info);
+	void		(*getUnicodeVersion) (UVersionInfo into);
+	void		(*getCLDRVersion) (UVersionInfo info, UErrorCode *status);
+	UCollator  *(*open) (const char *loc, UErrorCode *status);
+	void		(*close) (UCollator *coll);
+	void		(*getCollatorVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*getUCAVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*versionToString) (const UVersionInfo versionArray,
+									char *versionString);
+	UCollationResult (*strcoll) (const UCollator *coll,
+								 const UChar *source,
+								 int32_t sourceLength,
+								 const UChar *target,
+								 int32_t targetLength);
+	UCollationResult (*strcollUTF8) (const UCollator *coll,
+									 const char *source,
+									 int32_t sourceLength,
+									 const char *target,
+									 int32_t targetLength,
+									 UErrorCode *status);
+	int32_t		(*getSortKey) (const UCollator *coll,
+							   const UChar *source,
+							   int32_t sourceLength,
+							   uint8_t *result,
+							   int32_t resultLength);
+	int32_t		(*nextSortKeyPart) (const UCollator *coll,
+									UCharIterator *iter,
+									uint32_t state[2],
+									uint8_t *dest,
+									int32_t count,
+									UErrorCode *status);
+	void		(*setUTF8) (UCharIterator *iter,
+							const char *s,
+							int32_t length);
+	const char *(*errorName) (UErrorCode code);
+	int32_t		(*strToUpper) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToLower) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToTitle) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UBreakIterator *titleIter,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	void		(*setAttribute) (UCollator *coll,
+								 UColAttribute attr,
+								 UColAttributeValue value,
+								 UErrorCode *status);
+	struct pg_icu_library *next;
+} pg_icu_library;
+
+#endif
 
 /*
  * We define our own wrapper around locale_t so we can keep the same
@@ -84,17 +160,24 @@ struct pg_locale_struct
 		{
 			const char *locale;
 			UCollator  *ucol;
+			pg_icu_library *lib;
 		}			icu;
 #endif
 		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
 	}			info;
 };
 
+#ifdef USE_ICU
+#define PG_ICU_LIB(x) ((x)->info.icu.lib)
+#define PG_ICU_COL(x) ((x)->info.icu.ucol)
+#endif
+
 typedef struct pg_locale_struct *pg_locale_t;
 
 extern PGDLLIMPORT struct pg_locale_struct default_locale;
 
-extern void make_icu_collator(const char *iculocstr,
+extern bool make_icu_collator(const char *iculocstr,
+							  const char *collversion,
 							  struct pg_locale_struct *resultp);
 
 extern pg_locale_t pg_newlocale_from_collation(Oid collid);
diff --git a/src/test/icu/meson.build b/src/test/icu/meson.build
index 5a4f53f37f..ac2672190e 100644
--- a/src/test/icu/meson.build
+++ b/src/test/icu/meson.build
@@ -5,6 +5,7 @@ tests += {
   'tap': {
     'tests': [
       't/010_database.pl',
+      't/020_multiversion.pl',
     ],
     'env': {'with_icu': icu.found() ? 'yes' : 'no'},
   },
diff --git a/src/test/icu/t/020_multiversion.pl b/src/test/icu/t/020_multiversion.pl
new file mode 100644
index 0000000000..c04df4c65d
--- /dev/null
+++ b/src/test/icu/t/020_multiversion.pl
@@ -0,0 +1,274 @@
+# Copyright (c) 2022, PostgreSQL Global Development Group
+#
+# If one or more extra ICU versions is installed in the standard system library
+# search path, this test will detect them and run.
+
+use strict;
+use warnings;
+use PostgreSQL::Test::Cluster;
+use PostgreSQL::Test::Utils;
+use Test::More;
+
+if ($ENV{with_icu} ne 'yes')
+{
+	plan skip_all => 'ICU not supported by this build';
+}
+
+my $node1 = PostgreSQL::Test::Cluster->new('node1');
+$node1->init;
+$node1->start;
+
+# Check which ICU versions are installed.
+my $highest_version = $node1->safe_psql('postgres', 'select max(icu_version::decimal) from pg_icu_library_versions()');
+my $lowest_version = $node1->safe_psql('postgres', 'select min(icu_version::decimal) from pg_icu_library_versions()');
+my $highest_major_version = int($highest_version);
+my $lowest_major_version = int($lowest_version);
+
+if ($highest_major_version == $lowest_major_version)
+{
+	$node1->stop;
+	plan skip_all => 'no extra ICU library versions found';
+}
+
+sub set_default_icu_library_version
+{
+	my $icu_version = shift;
+	$node1->safe_psql('postgres', "alter system set default_icu_library_version = '$icu_version'; select pg_reload_conf()");
+}
+
+sub set_icu_library_versions
+{
+	my $icu_versions = shift;
+	$node1->safe_psql('postgres', "alter system set icu_library_versions = '$icu_versions'");
+	$node1->restart;
+}
+
+my $ret;
+my $stderr;
+
+# === DATABASE objects ===
+
+# ===== Scenario 1: user creates database with all default settings
+
+$node1->safe_psql('postgres', "create database db2 locale_provider = icu template = template0 icu_locale = 'en'");
+
+# No warning when logging into this database.
+$ret = $node1->psql('db2', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# ===== Scenario 2: user wants to use an old library
+
+# Create a database using the older library by changing the default.  This
+# might be done for compatibility with some other system, but it also simulates
+# a database that was created with all default settings when the binary was
+# linked against the older version.
+set_default_icu_library_version($lowest_major_version);
+$node1->safe_psql('postgres', "create database db3 locale_provider = icu template = template0 icu_locale = 'en'");
+
+isnt($node1->safe_psql('postgres', "select datcollversion from pg_database where datname = 'db2'"),
+     $node1->safe_psql('postgres', "select datcollversion from pg_database where datname = 'db3'"),
+     'db2 and db3 should have different datcollversion');
+
+# ===== Scenario 3: user has the old library avaliable, is happy to keep using it
+
+# Unset the default ICU library version (meaning use the linked version for
+# newly created databases).  No warning, because we can still find that older
+# version via dlopen().  User can happily go on using that old version in this
+# database for the rest of time.
+set_default_icu_library_version("");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# ===== Scenario 4: user doesn't have the old library, installs after warnings
+
+# Hide the old library version.  This simulates a system that doesn't have that
+# version installed yet, by making it unavailable.  We get a warning.
+set_icu_library_versions("$highest_major_version");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING:  database "db3" has a collation version mismatch/, "warning for incorrect datcollversion");
+like($stderr, qr/HINT:  Install a version of ICU that provides/, "warning suggests installing another ICU version");
+
+# Make the old version available again, this time explicitly (whereas before it
+# worked becuase the default is * which would find it automatically).
+set_icu_library_versions("$lowest_major_version,$highest_major_version");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# It also works if you use a major.minor version explicitly.
+set_icu_library_versions("$lowest_version,$highest_major_version");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# Or *, the default value that we started with.
+set_icu_library_versions("*");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# ===== Scenario 5: user doesn't have the old library, rebuilds after warnings
+
+# Hide the old library version again, and we get the warning again.
+set_icu_library_versions("$highest_major_version");
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING:  database "db3" has a collation version mismatch/, "warning for incorrect datcollversion");
+like($stderr, qr/HINT:  Install a version of ICU that provides/, "warning suggests installing another ICU version");
+
+# If we don't want to install a new library, we have the option of clobbering
+# the version.  It's the administrator's job to rebuild any database objects
+# that depend on the collation (most interestingly indexes) before doing so.
+# In this scenario, the REFRESH command can be run before or *after* rebuilding
+# indexes, because either way we're already using the default ICU library (due
+# to failure to find the named version).
+$ret = $node1->psql('postgres', "alter database db3 refresh collation version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+
+# Now no warning.
+$ret = $node1->psql('db3', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning after refresh");
+
+# ===== Scenario 6: user has the old library, but eventually decides to rebuild/upgrade
+
+# Make a new database with the old version active
+set_default_icu_library_version($lowest_major_version);
+$node1->safe_psql('postgres', "create database db4 locale_provider = icu template = template0 icu_locale = 'en'");
+
+# No warning, it just load the old version.
+$ret = $node1->psql('db4', "select", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning after refresh");
+my $old_datcollversion = $node1->safe_psql('postgres', "select datcollversion from pg_database where datname = 'db4'");
+
+# The user would now like to upgrade to the new library.  Presumably people
+# will want to do this eventually to avoid running very old unmaintained copies
+# of ICU.  Unlike scenario 3, here it's actually a requirement to REFRESH
+# *before* doing all the rebuilds of indexes etc, which may be a little
+# confusing (not shown here).  REFRESH is necessary to change datcollversion,
+# which is required to make us start opening the newer library.
+#
+# XXX Currently you also need to reconnect all sessions too, because the
+# default locale is cached and now out of date.
+set_default_icu_library_version("");
+$ret = $node1->psql('postgres', "alter database db4 refresh collation version", stderr => \$stderr);
+my $new_datcollversion = $node1->safe_psql('postgres', "select datcollversion from pg_database where datname = 'db4'");
+
+isnt($old_datcollversion, $new_datcollversion, "datcollversion changed");
+
+
+# === COLLATION objects ===
+
+# The same scenarios, this time with COLLATIONs.
+
+# ===== Scenario 1: user creates database with all default settings
+
+set_default_icu_library_version("");
+set_icu_library_versions("*");
+$node1->safe_psql('postgres', "create collation c1 (provider = icu, locale = 'en')");
+
+# No warning when using it.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c1", stderr => \$stderr);
+is($ret, 0, "can use collation");
+unlike($stderr, qr/WARNING/, "no warning for default");
+
+# ===== Scenario 2: user wants to use an old library
+
+# Simulates a collation in a database that migrated from an older binary, or a
+# collation set up explicitly to match some other system.
+set_default_icu_library_version($lowest_major_version);
+$node1->safe_psql('postgres', "create collation c2 (provider = icu, locale = 'en')");
+
+isnt($node1->safe_psql('postgres', "select collversion from pg_collation where collname = 'c1'"),
+     $node1->safe_psql('postgres', "select collversion from pg_collation where collname = 'c2'"),
+     'c1 and c2 should have different collversion');
+
+# ===== Scenario 3: user has the old library avaliable, is happy to keep using it
+
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "can use collation");
+unlike($stderr, qr/WARNING/, "no warning when using old library collation");
+
+# ===== Scenario 4: user doesn't have the old library, installs after warnings
+
+# Hide the old library version.
+set_default_icu_library_version("");
+set_icu_library_versions("$highest_major_version");
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING:  collation "c2" version mismatch/, "warning for incorrect collversion");
+like($stderr, qr/HINT:  Install a version of ICU that provides/, "warning suggests installing another ICU version");
+
+# Make the old version available again.
+set_icu_library_versions("$lowest_major_version,$highest_major_version");
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# It also works if you use a major.minor version explicitly.
+set_icu_library_versions("$lowest_version,$highest_major_version");
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# Or *, the default value that we started with.
+set_icu_library_versions("*");
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# ===== Scenario 5: user doesn't have the old library, rebuilds after warnings
+
+# Hide the old library version again.
+set_icu_library_versions("$highest_major_version");
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/WARNING:  collation "c2" version mismatch/, "warning for incorrect collversion");
+like($stderr, qr/HINT:  Install a version of ICU that provides/, "warning suggests installing another ICU version");
+
+# Rebuild things, and refresh.
+$ret = $node1->psql('postgres', "alter collation c2 refresh version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+
+# Now no warning.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c2", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+# ===== Scenario 6: user has the old library, but eventually decides to rebuild/upgrade
+
+set_default_icu_library_version($lowest_major_version);
+$node1->safe_psql('postgres', "create collation c3 (provider = icu, locale = 'en')");
+
+# No warning.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c3", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+my $old_collversion = $node1->safe_psql('postgres', "select collversion from pg_collation where collname = 'c3'");
+
+# Rebuild things, and refresh.  As with database scenario 6, we need to refresh
+# *before* rebuilding dependent objects (not shown here).
+set_default_icu_library_version("");
+$ret = $node1->psql('postgres', "alter collation c3 refresh version", stderr => \$stderr);
+is($ret, 0, "success");
+like($stderr, qr/NOTICE:  changing version/, "version changes");
+
+# No warning.
+$ret = $node1->psql('postgres', "select 'x' < 'y' collate c3", stderr => \$stderr);
+is($ret, 0, "success");
+unlike($stderr, qr/WARNING/, "no warning");
+
+my $new_collversion = $node1->safe_psql('postgres', "select collversion from pg_collation where collname = 'c3'");
+
+isnt($old_collversion, $new_collversion, "collversion changed");
+
+$node1->stop;
+
+done_testing();
diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list
index 2f5802195d..50d9558cf4 100644
--- a/src/tools/pgindent/typedefs.list
+++ b/src/tools/pgindent/typedefs.list
@@ -1101,6 +1101,7 @@ HeapTupleTableSlot
 HistControl
 HotStandbyState
 I32
+ICU_Convert_BI_Func
 ICU_Convert_Func
 ID
 INFIX
@@ -2852,10 +2853,12 @@ TypeName
 U
 U32
 U8
+UBreakIterator
 UChar
 UCharIterator
 UColAttribute
 UColAttributeValue
+UCollationResult
 UCollator
 UConverter
 UErrorCode
@@ -3482,6 +3485,7 @@ pg_funcptr_t
 pg_gssinfo
 pg_hmac_ctx
 pg_hmac_errno
+pg_icu_library
 pg_int64
 pg_local_to_utf_combined
 pg_locale_t
-- 
2.38.1



  [text/x-patch] v8-0002-ci-XXX-install-ICU63-on-debian.patch (929B, ../../CA+hUKGLr=9d+-k8PVv8e__TOxuq=n0SKNDpqCzbGrK5EbDyxAg@mail.gmail.com/3-v8-0002-ci-XXX-install-ICU63-on-debian.patch)
  download | inline diff:
From aa9ac4f827f83abd4efe7f773efa2e2f45ad7640 Mon Sep 17 00:00:00 2001
From: Thomas Munro <[email protected]>
Date: Sat, 26 Nov 2022 14:39:18 +1300
Subject: [PATCH v8 2/2] ci: XXX install ICU63 on debian

This is not a good way to add the package, just doing this temporarily
as a demonstration via cfbot.
---
 .cirrus.yml | 4 ++++
 1 file changed, 4 insertions(+)

diff --git a/.cirrus.yml b/.cirrus.yml
index f31923333e..8c1fb63cad 100644
--- a/.cirrus.yml
+++ b/.cirrus.yml
@@ -296,6 +296,10 @@ task:
   setup_additional_packages_script: |
     #apt-get update
     #DEBIAN_FRONTEND=noninteractive apt-get -y install ...
+    # this is debian 11 (bullseye) but we can install ICU 63 from debian 10 (buster)
+    curl -O http://ftp.debian.org/debian/pool/main/i/icu/libicu63_63.1-6+deb10u3_amd64.deb
+    dpkg -i libicu63_63.1-6+deb10u3_amd64.deb
+
 
   matrix:
     - name: Linux - Debian Bullseye - Autoconf
-- 
2.38.1



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-28 06:10  Thomas Munro <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-28 06:10 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sat, Nov 26, 2022 at 6:27 PM Thomas Munro <[email protected]> wrote:
> This is just a first cut, but enough to try out and see if we like it,
> what needs to be improved, what edge cases we haven't thought about
> etc.  Let me know what you think.

BTW one problem to highlight (mentioned but buried in the test
comments), is that REFRESH VERSION doesn't affect other sessions or
even the current session.  You have to log out and back in again to
pick up the new version.  Obviously that's not good enough, but fixing
that involves making it transactional, I think.  If you abort, we have
to go back to using the old version, if you commit you keep the new
version and we might also consider telling other backends to start
using the new version -- or something like that.  I think that's just
a Small Matter of Programming, but a little bit finickity and I need
to take a break for a bit and go work on bugs elsewhere, hence v8
didn't address that yet.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-28 19:11  Robert Haas <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 2 replies; 57+ messages in thread

From: Robert Haas @ 2022-11-28 19:11 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeff Davis <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 23, 2022 at 12:09 AM Thomas Munro <[email protected]> wrote:
> OK.  Time for a new list of the various models we've discussed so far:
>
> 1.  search-by-collversion:  We introduce no new "library version"
> concept to COLLATION and DATABASE object and little or no new syntax.
>
> 2.  lib-version-in-providers: We introduce a separate provider value
> for each ICU version, for example ICU63, plus an unversioned ICU like
> today.
>
> 3.  lib-version-in-attributes: We introduce daticuversion (alongside
> datcollversion) and collicuversion (alongside collversion).  Similar
> to the above, but it's a separate property and the provider is always
> ICU.  New syntax for CREATE/ALTER COLLATION/DATABASE to set and change
> ICU_VERSION.
>
> 4.  lib-version-in-locale:  "63:en" from earlier versions.  That was
> mostly a strawman proposal to avoid getting bogged down in
> syntax/catalogue/model change discussions while trying to prove that
> dlopen would even work.  It doesn't sound like anyone really likes
> this.
>
> 5.  lib-version-in-collversion:  We didn't explicitly discuss this
> before, but you hinted at it: we could just use u_getVersion() in
> [dat]collversion.

I'd like to vote against #3 at least in the form that's described
here. If we had three more libraries providing collations, it's likely
that they would need versioning, too. So if we add an explicit notion
of provider version, then it ought not to be specific to libicu.

I think it's OK to decide that different library versions are
different providers (your option #2), or that they are the same
provider but give rise to different collations (your option #4), or
that there can be multiple version of each collation which are
distinguished by some additional provider version field (your #3 made
more generic).

I don't really understand #1 or #5 well enough to have an educated
opinion, but I do think that #1 seems a bit magical. It hopes that the
combination of a collation name and a datcollversion will be
sufficient to find exactly one matcing collation in a list of provided
libraries. The advantage of that, as I understand it, is that if you
do something to your system that causes the number of matches to go
from one to zero, you can just throw another library on the pile and
get the number back up to one. Woohoo! But there's a part of me that
worries: what if the number goes up to two, and they're not all the
same? Probably that's something that shouldn't happen, but if it does
then I think there's kind of no way to fix it. With the other options,
if there's some way to jigger the catalog state to match what you want
to happen, you can always repair the situation somehow, because the
library to be used for each collation is explicitly specified in some
way, and you just have to get it to match what you want to have
happen.

I don't know too much about this, though, so I might have it all wrong.

-- 
Robert Haas
EDB: http://www.enterprisedb.com





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 02:54  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 3 replies; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 02:54 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sat, 2022-11-26 at 18:27 +1300, Thomas Munro wrote:
> Here's the first iteration.

I will send a full review shortly, but I encountered an ICU bug along
the way, which caused me some confusion for a bit. I'll skip past the
various levels of confusion I had (burned a couple hours), and get
right to the repro:

Install the latest release of all major versions 50-69, and compile
postgres against 70. You'll get:

=# select * from pg_icu_collation_versions('en_US') order by
icu_version;
 icu_version | uca_version | collator_version 
-------------+-------------+------------------
 50.2        | 6.2         | 58.0.6.50
 51.3        | 6.2         | 58.0.6.50
 52.2        | 6.2         | 58.0.6.50
 53.2        | 6.3         | 137.51
 54.2        | 7.0         | 137.56
 55.2        | 7.0         | 153.56
 56.2        | 8.0         | 153.64
 57.2        | 8.0         | 153.64
 58.3        | 9.0         | 153.72
 59.2        | 9.0         | 153.72
 60.3        | 10.0        | 153.80
 61.2        | 10.0        | 153.80
 62.2        | 11.0        | 153.88
 63.2        | 11.0        | 153.88
 64.2        | 12.1        | 153.97
 65.1        | 12.1        | 153.97
 66.1        | 13.0        | 153.14
 67.1        | 13.0        | 153.14
 68.2        | 13.0        | 153.14
 69.1        | 13.0        | 153.14
 70.1        | 14.0        | 153.112
(21 rows)

This is good information, because it tells us that major library
versions change more often than collation versions, empirically-
speaking.

But did you notice that the version went backwards from 65.1 -> 66.1?
Well, actually, it didn't. The version of that collation in 66.1 went
from 153.97 -> 153.104. But there's a bug in versionToString() that
does the decimal output incorrectly when there's a '0' digit between
the hundreds and the ones place. I'll see about reporting that, but I
thought I'd mention it here because it could have consequences, as we
are storing the strings :-(

The bug is still present in 70.1, but it's masked because it went to
.112.

Incidentally, this answers our other question about whether the
collation version can change in a minor version update. Perhaps not,
but if they fix this bug and backport it, then the version *string*
will change in a minor update. Ugh.

Regards,
	Jeff Davis






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 02:57  Robert Haas <[email protected]>
  parent: Jeff Davis <[email protected]>
  2 siblings, 1 reply; 57+ messages in thread

From: Robert Haas @ 2022-11-29 02:57 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, Nov 28, 2022 at 9:55 PM Jeff Davis <[email protected]> wrote:
> But did you notice that the version went backwards from 65.1 -> 66.1?
> Well, actually, it didn't. The version of that collation in 66.1 went
> from 153.97 -> 153.104. But there's a bug in versionToString() that
> does the decimal output incorrectly when there's a '0' digit between
> the hundreds and the ones place. I'll see about reporting that, but I
> thought I'd mention it here because it could have consequences, as we
> are storing the strings :-(
>
> The bug is still present in 70.1, but it's masked because it went to
> .112.
>
> Incidentally, this answers our other question about whether the
> collation version can change in a minor version update. Perhaps not,
> but if they fix this bug and backport it, then the version *string*
> will change in a minor update. Ugh.

That is ... astonishingly bad.

-- 
Robert Haas
EDB: http://www.enterprisedb.com





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 03:36  Jeff Davis <[email protected]>
  parent: Robert Haas <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 03:36 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, 2022-11-28 at 21:57 -0500, Robert Haas wrote:
> That is ... astonishingly bad.

https://unicode-org.atlassian.net/browse/CLDR-16175


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 04:34  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  2 siblings, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 04:34 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 29, 2022 at 3:55 PM Jeff Davis <[email protected]> wrote:
> =# select * from pg_icu_collation_versions('en_US') order by
> icu_version;
>  icu_version | uca_version | collator_version
> -------------+-------------+------------------
>  50.2        | 6.2         | 58.0.6.50
>  51.3        | 6.2         | 58.0.6.50
>  52.2        | 6.2         | 58.0.6.50
>  53.2        | 6.3         | 137.51
>  54.2        | 7.0         | 137.56
>  55.2        | 7.0         | 153.56
>  56.2        | 8.0         | 153.64
>  57.2        | 8.0         | 153.64
>  58.3        | 9.0         | 153.72
>  59.2        | 9.0         | 153.72
>  60.3        | 10.0        | 153.80
>  61.2        | 10.0        | 153.80
>  62.2        | 11.0        | 153.88
>  63.2        | 11.0        | 153.88
>  64.2        | 12.1        | 153.97
>  65.1        | 12.1        | 153.97
>  66.1        | 13.0        | 153.14
>  67.1        | 13.0        | 153.14
>  68.2        | 13.0        | 153.14
>  69.1        | 13.0        | 153.14
>  70.1        | 14.0        | 153.112
> (21 rows)
>
> This is good information, because it tells us that major library
> versions change more often than collation versions, empirically-
> speaking.

Wow, nice discovery about 104 -> 14.  Yeah, I imagine we'll want some
kind of band-aid to tolerate that exact screwup and avoid spurious
warnings.

Bugs aside, that's quite a revealing table in other ways.  We can see:

* The version scheme changed completely in ICU 53.  This corresponds
to a major rewrite of the collation code, I see[1].

* The first component seems to be (UCOL_RUNTIME_VERSION << 4) + 9.
UCOL_RUNTIME_VERSION is in their uvernum.h, currently 9, was 8, bumped
between 54 and 55 (I see this in their commit log), corresponding to
the two possible numbers 137 and 153 that we see there.  I don't know
where the final 9 term is coming from but it looks stable since the v2
collation rewrite landed.

* The second component seems to be uca_version_major * 8 +
uca_version_minor (that's the Unicode Collation Algorithm version, and
so far always matches the Unicode version, visible in the output of
the other function).

* The values you showed for English don't have a third component, but
if you try some other locales like 'zh' you'll see the CLDR major
version in third position.  So I guess some locales depend on CLDR
data and others don't.

TL;DR it *looks* like the set of ingredients for the version string is:

* UCOL_RUNTIME_VERSION (rarely changes)
* UCA/Unicode major.minor version
* sometimes CLDR major version, not sure when
* 9

[1] https://icu.unicode.org/design/collation/v2





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 04:48  Jeff Davis <[email protected]>
  parent: Robert Haas <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 04:48 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, 2022-11-28 at 14:11 -0500, Robert Haas wrote:
> I don't really understand #1 or #5 well enough to have an educated
> opinion, but I do think that #1 seems a bit magical. It hopes that
> the
> combination of a collation name and a datcollversion will be
> sufficient to find exactly one matcing collation in a list of
> provided
> libraries. The advantage of that, as I understand it, is that if you
> do something to your system that causes the number of matches to go
> from one to zero, you can just throw another library on the pile and
> get the number back up to one. Woohoo! But there's a part of me that
> worries: what if the number goes up to two, and they're not all the
> same? Probably that's something that shouldn't happen, but if it does
> then I think there's kind of no way to fix it. With the other
> options,
> if there's some way to jigger the catalog state to match what you
> want
> to happen, you can always repair the situation somehow, because the
> library to be used for each collation is explicitly specified in some
> way, and you just have to get it to match what you want to have
> happen.

Not necessarily, #2-4 (at least as implemented in v7) can only load one
major version at a time, so can't specify minor versions:
https://www.postgresql.org/message-id/[email protected]

With #1, you can provide control over the search order to find the
symbol you want. Granted, if you want to specify that different
collations look in different libraries for the same version, then it
won't work, because the search order is global -- is that what you're
worried about? If so, I think we need to compare it against the
downsides of #2-4, which in my opinion are more serious.

The first thing to sort out with options #2-4 is: what about minor
versions? V7 took the approach that only the major version matters.
That means that if you want to select a specific minor version, then
you are out of luck, because only one major at a time can be loaded,
globally. But paying attention to minor versions seems like a mess --
we'd need even more magical fallbacks that try later minor versions or
something.

Second, there is weirdness in the common case that a collation version
doesn't change between versions. Let's say you have a collation
"mycoll" with locale "en_US" and it's pointed at built-in library
version 64, with collation version 153.97. GUC
default_icu_library_version is set to 63. Then you upgrade the system
and ICU gets updated from 64 -> 65. Now, it can't find version 64 to
load, so it falls back to 63 (which has the wrong version 153.88), even
though 65 is just fine because it still offers that locale with version
153.97. (A similar problem exists when you remove a version of ICU from
icu_library_path, and another version suffices for all of your
collations.)

Thirdly, as I said earlier, it's just hard on the user to try to sort
out two different versions modeled in the database. Understanding
encodings and collations are hard enough, and then we introduce *two*
versions on top of that.

Fourth, I don't see what the point of ucol_getVersion() is in schemes
#2-4. All it does is control a WARNING, because throwing an error (at
least by default) would be too harsh, given that users have lived with
these risks for so long. But if all it does is throw a warning, what's
the point in modeling it in the catalog as though it's the most
important version?

Ultimately, I think collation version (as reported by
ucol_getVersion()) is the most accurate and least-surprising way to
match a library-provided collation with the collation in the catalog.
And it seems like we'd be using it in exactly the way the ICU
maintainers intend it to be used.

Of course, I cast my vote for #1 before I discovered this ICU bug
here: 
https://www.postgresql.org/message-id/[email protected]

That injects some doubt, to be sure. If I were to try to solve the
problems with #2-4, one approach might be to treat the built-in ICU
version differently from the ones in icu_library_path. Not quite sure,
I'd have to think more. But as of now, I'd still lean toward #1 until a
better option is presented.

Regards,
	Jeff Davis






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 06:51  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  2 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 06:51 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sat, 2022-11-26 at 18:27 +1300, Thomas Munro wrote:
> On Thu, Nov 24, 2022 at 5:48 PM Thomas Munro <[email protected]>
> wrote:
> > On Thu, Nov 24, 2022 at 3:07 PM Jeff Davis <[email protected]>
> > wrote:
> > > I'd vote for 1 on the grounds that it's easier to document and
> > > understand a single collation version, which comes straight from
> > > ucol_getVersion(). This approach makes it a separate problem to
> > > find
> > > the collation version among whatever libraries the admin can
> > > provide;
> > > but adding some observability into the search should mitigate any
> > > confusion.
> > 
> > OK, it sounds like I should code that up next.
> 
> Here's the first iteration.

Thank you.

Proposed changes:

* I attached a first pass of some documentation.

* Should be another GUC to turn WARNING into an ERROR. Useful at least
for testing; perhaps too dangerous for production.

* The libraries should be loaded in a more diliberate order. The "*"
should be expanded in a descending fashion so that later versions are
preferred.

* GUCs should be validated.

* Should validate that loaded library has expected version.

* We need to revise or remove pg_collation_actual_version() and
pg_database_collation_actual_version().

* The GUCs are PGC_SUSET, but don't take effect because
icu_library_list_fully_loaded is never reset.

* The extra collations you're adding at bootstrap time are named based
on the library major version. I suppose it might be more "proper" to
name them based on the collation version, but that would be more
verbose, so I won't advocate for that. Just pointing it out.

* It looks hard (or impossible) to mix multiple ICU libraries with the
same major version and different minor versions. That's because,
e.g., libicui18n.so.63.1 links against libicuuc.63 and libicudata.63,
and when you install ICU 63.2, those dependencies get clobbered with
the 63.2 versions. That fails the sanity check I proposed above about
the library version number matching the requested library version
number. And it also just seems wrong -- why would you have minor-
version precision about an ICU library but then only major-version
precision about the ICU dependencies of that library? Doesn't that
defeat the whole purpose of this naming scheme? (Maybe another ICU
bug?).

Minor comments:

* ICU_I18N is defined in make_icu_library_name() but used outside of
it. One solution might be to have it return both library names to the
caller and rename it as make_icu_library_names().

* get_icu_function() could use a clarifying comment or a better name.
Something that communicates that you are looking for the function in
the given library with the given major version number (which may or may
not be needed depending on how the library was compiled).

* typo in comment over make_icu_collator:
s/u_getVersion/ucol_getVersion/

* The return value of make_icu_collator() seems backwards to me,
stylistically. I typically see the false-is-good pattern with integer
returns.

* weird bracketing style in get_icu_collator for the "else"

>   The version rosetta stone functions look like this:
> 
> postgres=# select * from pg_icu_library_versions();
>  icu_version | unicode_version | cldr_version
> -------------+-----------------+--------------
>  67.1        | 13.0            | 37.0
>  63.1        | 11.0            | 34.0
>  57.1        | 8.0             | 29.0
> (3 rows)
> 
> postgres=# select * from pg_icu_collation_versions('zh');
>  icu_version | uca_version | collator_version
> -------------+-------------+------------------
>  67.1        | 13.0        | 153.14.37
>  63.1        | 11.0        | 153.88.34
>  57.1        | 8.0         | 153.64.29
> (3 rows)

I like these functions.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS




Attachments:

  [text/x-patch] doc.patch (8.3K, ../../[email protected]/2-doc.patch)
  download | inline diff:
diff --git a/doc/src/sgml/charset.sgml b/doc/src/sgml/charset.sgml
index 445fd175d8..b9dba8ac67 100644
--- a/doc/src/sgml/charset.sgml
+++ b/doc/src/sgml/charset.sgml
@@ -1047,6 +1047,50 @@ CREATE COLLATION ignore_accents (provider = icu, locale = 'und-u-ks-level1-kc-tr
     </tip>
    </sect3>
   </sect2>
+  <sect2 id="collation-versions">
+   <title>Collation Versions</title>
+
+   <para>
+    Collations are sensitive to the specific collation version, which is
+    obtained from the collation provider library at the time the collation is
+    created (and only updated with <xref linkend="sql-altercollation"/>). If
+    the collation provider library is updated on the system (e.g. due to an
+    operating system upgrade), it may provide a different collation version;
+    but the version recorded in <productname>PostgreSQL</productname> will
+    remain unchanged.
+   </para>
+   <para>
+    New collation versions are generally desirable, as they reflect changes in
+    natural language over time. But these ordering changes can also cause
+    problems, such as the inconsistency of an indexes, which often depend on a
+    stable ordering. If <productname>PostgreSQL</productname> is unable to
+    find a collation in the collation provider that matches the recorded
+    version exactly, it will emit a <literal>WARNING</literal> (configurable
+    with <xref linkend="guc-collation-version-mismatch-error"/>).
+   </para>
+   <sect3 id="multiple-icu-libraries">
+    <title>Multiple ICU collation provider libraries</title>
+    <para>
+     When using the <literal>icu</literal> collation provider,
+     <productname>PostgreSQL</productname> can load multiple ICU collation
+     provider libraries, making it possible to find an exact match for the
+     collation version even if the operating system's ICU library has been
+     upgraded and provides a new collation version.
+    </para>
+    <para>
+     To use additional ICU collation provider libraries, set <xref
+     linkend="guc-icu-library-path"/> to the directory where the ICU libraries
+     are installed, and use <xref linkend="guc-icu-library-versions"/> to
+     control how those libraries are searched.
+    </para>
+    <note>
+     <para>
+      The <literal>libc</literal> collation provider does not allow specifying
+      multiple libraries.
+     </para>
+    </note>
+   </sect3>
+  </sect2>
  </sect1>
 
  <sect1 id="multibyte">
diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml
index 9fd2075b1e..3809c26b31 100644
--- a/doc/src/sgml/config.sgml
+++ b/doc/src/sgml/config.sgml
@@ -10288,6 +10288,107 @@ dynamic_library_path = 'C:\tools\postgresql;H:\my_project\lib;$libdir'
      </variablelist>
     </sect2>
 
+    <sect2 id="runtime-config-compatible-collation">
+     <title>Collation Version Compatibility</title>
+     <variablelist>
+     <varlistentry id="guc-collation-version-mismatch-error">
+      <term><varname>collation_version_mismatch_error</varname> (<type>boolean</type>)
+      <indexterm>
+       <primary><varname>collation_version_mismatch_error</varname> configuration parameter</primary>
+      </indexterm>
+      </term>
+      <listitem>
+       <para>
+        If <productname>PostgreSQL</productname> detects mismatched collation
+        versions, and this variable is set to <literal>true</literal>, an
+        error will be raised. If this variable is set to
+        <literal>false</literal>, a warning will be raised instead. The
+        default is <literal>false</literal>.
+       </para>
+       <para>
+        See <xref linkend="collation-versions"/> for more information.
+       </para>
+      </listitem>
+     </varlistentry>
+
+     <varlistentry id="guc-icu-library-path">
+      <term><varname>icu_library_path</varname> (<type>string</type>)
+      <indexterm>
+       <primary><varname>icu_library_path</varname> configuration parameter</primary>
+      </indexterm>
+      </term>
+      <listitem>
+       <para>
+        Set to the directory where additional ICU libraries are installed, to
+        be searched for matching collation versions.
+       </para>
+       <para>
+        See <xref linkend="multiple-icu-libraries"/> for more information.
+       </para>
+      </listitem>
+     </varlistentry>
+
+     <varlistentry id="guc-icu-library-versions">
+      <term><varname>icu_library_versions</varname> (<type>string</type>)
+      <indexterm>
+       <primary><varname>icu_library_versions</varname> configuration parameter</primary>
+      </indexterm>
+      </term>
+      <listitem>
+       <para>
+        When searching for a matching collation version, search the ICU
+        libraries with the version numbers specified in this setting,
+        separated by commas. By default, only the built-in ICU library is
+        searched.
+       </para>
+       <para>
+        Library version numbers can be specified as either
+        <literal>major_version</literal> or
+        <literal>major_version.minor_version</literal>. By default, the
+        built-in ICU library is used.
+       </para>
+       <para>
+        If this variable is set to <literal>*</literal>,
+        <productname>PostgreSQL</productname> will attempt to load any ICU
+        library in <literal>icu_library_path</literal> with a compatible major
+        version.
+       </para>
+       <para>
+        See <xref linkend="multiple-icu-libraries"/> for more information.
+       </para>
+      </listitem>
+     </varlistentry>
+
+     <varlistentry id="guc-default-icu-library-version">
+      <term><varname>default_icu_library_version</varname> (<type>string</type>)
+      <indexterm>
+       <primary><varname>default_icu_library_version</varname> configuration parameter</primary>
+      </indexterm>
+      </term>
+      <listitem>
+       <para>
+        If <productname>PostgreSQL</productname> detects mismatched collation
+        versions, use the collation provided by the ICU library with this
+        version number.
+       </para>
+       <para>
+        Library version numbers can be specified as either
+        <literal>major_version</literal> or
+        <literal>major_version.minor_version</literal>. By default, the
+        built-in ICU library is used.
+       </para>
+       <para>
+        Has no effect if <literal>collation_version_mismatch_error</literal>
+        is set to <literal>true</literal>.
+       </para>
+       <para>
+        See <xref linkend="multiple-icu-libraries"/> for more information.
+       </para>
+      </listitem>
+     </varlistentry>
+
+     </variablelist>
+    </sect2>
     <sect2 id="runtime-config-compatible-clients">
      <title>Platform and Client Compatibility</title>
      <variablelist>
diff --git a/doc/src/sgml/func.sgml b/doc/src/sgml/func.sgml
index 68cd4297d2..a9f6258e77 100644
--- a/doc/src/sgml/func.sgml
+++ b/doc/src/sgml/func.sgml
@@ -27180,6 +27180,39 @@ postgres=# SELECT * FROM pg_walfile_name_offset((pg_backup_stop()).lsn);
         Use of this function is restricted to superusers.
        </para></entry>
       </row>
+
+      <row>
+       <entry role="func_table_entry"><para role="func_signature">
+        <indexterm>
+         <primary>pg_icu_library_versions</primary>
+        </indexterm>
+        <function>pg_icu_library_versions</function> ()
+        <returnvalue>setof record</returnvalue>
+        ( <parameter>icu_version</parameter> <type>text</type>,
+        <parameter>unicode_version</parameter> <type>text</type>,
+        <parameter>cldr_version</parameter> <type>text</type>) )
+       </para>
+       <para>
+        Returns information from each available ICU library.
+       </para></entry>
+      </row>
+
+      <row>
+       <entry role="func_table_entry"><para role="func_signature">
+        <indexterm>
+         <primary>pg_icu_collation_versions</primary>
+        </indexterm>
+        <function>pg_icu_collation_versions</function> ( <parameter>locale</parameter> <type>text</type> )
+        <returnvalue>setof record</returnvalue>
+        (<parameter>icu_version</parameter> <type>text</type>,
+        <parameter>uca_version</parameter> <type>text</type>,
+        <parameter>collator_version</parameter> <type>text</type> )
+       </para>
+       <para>
+        Returns the collation version of the given locale from each available
+        ICU library.
+       </para></entry>
+      </row>
      </tbody>
     </tgroup>
    </table>


^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 16:27  Joe Conway <[email protected]>
  parent: Robert Haas <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Joe Conway @ 2022-11-29 16:27 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; Thomas Munro <[email protected]>; +Cc: Jeff Davis <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 11/28/22 14:11, Robert Haas wrote:
> On Wed, Nov 23, 2022 at 12:09 AM Thomas Munro <[email protected]> wrote:
>> OK.  Time for a new list of the various models we've discussed so far:
>>
>> 1.  search-by-collversion:  We introduce no new "library version"
>> concept to COLLATION and DATABASE object and little or no new syntax.
>>
>> 2.  lib-version-in-providers: We introduce a separate provider value
>> for each ICU version, for example ICU63, plus an unversioned ICU like
>> today.
>>
>> 3.  lib-version-in-attributes: We introduce daticuversion (alongside
>> datcollversion) and collicuversion (alongside collversion).  Similar
>> to the above, but it's a separate property and the provider is always
>> ICU.  New syntax for CREATE/ALTER COLLATION/DATABASE to set and change
>> ICU_VERSION.
>>
>> 4.  lib-version-in-locale:  "63:en" from earlier versions.  That was
>> mostly a strawman proposal to avoid getting bogged down in
>> syntax/catalogue/model change discussions while trying to prove that
>> dlopen would even work.  It doesn't sound like anyone really likes
>> this.
>>
>> 5.  lib-version-in-collversion:  We didn't explicitly discuss this
>> before, but you hinted at it: we could just use u_getVersion() in
>> [dat]collversion.
> 
> I'd like to vote against #3 at least in the form that's described
> here. If we had three more libraries providing collations, it's likely
> that they would need versioning, too. So if we add an explicit notion
> of provider version, then it ought not to be specific to libicu.

+many

> I think it's OK to decide that different library versions are
> different providers (your option #2), or that they are the same
> provider but give rise to different collations (your option #4), or
> that there can be multiple version of each collation which are
> distinguished by some additional provider version field (your #3 made
> more generic).

I think provider and collation version are distinct concepts. The 
provider ('c' versus 'i' for example) determines a unique code path in 
the backend due to different APIs, whereas collation version is related 
to a specific ordering given a set of characters.


> I don't really understand #1 or #5 well enough to have an educated
> opinion, but I do think that #1 seems a bit magical. It hopes that the
> combination of a collation name and a datcollversion will be
> sufficient to find exactly one matcing collation in a list of provided
> libraries. The advantage of that, as I understand it, is that if you
> do something to your system that causes the number of matches to go
> from one to zero, you can just throw another library on the pile and
> get the number back up to one. Woohoo! But there's a part of me that
> worries: what if the number goes up to two, and they're not all the
> same? Probably that's something that shouldn't happen, but if it does
> then I think there's kind of no way to fix it. With the other options,
> if there's some way to jigger the catalog state to match what you want
> to happen, you can always repair the situation somehow, because the
> library to be used for each collation is explicitly specified in some
> way, and you just have to get it to match what you want to have
> happen.

My vote is for something like #5. The collversion should indicate a 
specific immutable ordering behavior.


-- 
Joe Conway
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 17:32  Robert Haas <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Robert Haas @ 2022-11-29 17:32 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, Nov 28, 2022 at 11:49 PM Jeff Davis <[email protected]> wrote:
> Not necessarily, #2-4 (at least as implemented in v7) can only load one
> major version at a time, so can't specify minor versions:
> https://www.postgresql.org/message-id/[email protected]
>
> With #1, you can provide control over the search order to find the
> symbol you want. Granted, if you want to specify that different
> collations look in different libraries for the same version, then it
> won't work, because the search order is global -- is that what you're
> worried about? If so, I think we need to compare it against the
> downsides of #2-4, which in my opinion are more serious.

You know more about this than I do, for sure, so don't let my vote
back the project into a bad spot. But, yeah, the thing you mention
here is what I'm worried about. Without a way to force a certain
behavior for a certain particular collation, you don't have an escape
valve if the global library ordering isn't doing what you want. Your
argument seems to at least partly be that #1 will be more usable on
the whole, and that does seem like an important consideration. People
may have a lot of collations and adjusting them all individually could
be difficult and unpleasant. However, I think it's also worth asking
what options someone has if #1 can't be made to work due to a single
ordering controlling every collation.

It's entirely possible that the scenario I'm worried about is too
remote in practice to be concerned about. I don't know how this stuff
works well enough to be certain. It's just that, on the basis of
previous experience, (1) it's not that uncommon for people to actually
end up in situations that we thought shouldn't ever happen and (2)
code that deals with collations is more untrustworthy than average.

-- 
Robert Haas
EDB: http://www.enterprisedb.com





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 18:03  Jeremy Schneider <[email protected]>
  parent: Jeff Davis <[email protected]>
  2 siblings, 1 reply; 57+ messages in thread

From: Jeremy Schneider @ 2022-11-29 18:03 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 11/28/22 6:54 PM, Jeff Davis wrote:

> 
> =# select * from pg_icu_collation_versions('en_US') order by
> icu_version;
>  icu_version | uca_version | collator_version 
> -------------+-------------+------------------
>  ...
>  67.1        | 13.0        | 153.14
>  68.2        | 13.0        | 153.14
>  69.1        | 13.0        | 153.14
>  70.1        | 14.0        | 153.112
> (21 rows)
> 
> This is good information, because it tells us that major library
> versions change more often than collation versions, empirically-
> speaking.


It seems to me that the collator_version field is not a good version
identifier to use.

Just taking a quick glance at the ICU home page right now, it shows that
all of the last 5 versions of ICU have included "additions and
corrections" to locale data itself, including 68 to 69 where the
collator version did not change.

Is it possible that this "collator_version" only reflects the code that
processes collation data to do comparisons/sorts, but it does not
reflect updates to the locale data itself?

https://icu.unicode.org/

ICU v72 -> CLDR v42
ICU v71 -> CLDR v41
ICU v70 -> CLDR v40
ICU v69 -> CLDR v39
ICU v68 -> CLDR v38

-Jeremy


-- 
http://about.me/jeremy_schneider






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 18:18  Thomas Munro <[email protected]>
  parent: Jeremy Schneider <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 18:18 UTC (permalink / raw)
  To: Jeremy Schneider <[email protected]>; +Cc: Jeff Davis <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 7:03 AM Jeremy Schneider
<[email protected]> wrote:
> It seems to me that the collator_version field is not a good version
> identifier to use.
>
> Just taking a quick glance at the ICU home page right now, it shows that
> all of the last 5 versions of ICU have included "additions and
> corrections" to locale data itself, including 68 to 69 where the
> collator version did not change.
>
> Is it possible that this "collator_version" only reflects the code that
> processes collation data to do comparisons/sorts, but it does not
> reflect updates to the locale data itself?

I think it also includes the CLDR version for *some* locales.  From a
quick look, that includes 'ar', 'ru', 'tr', 'zh'.  Jeff, would you
mind sharing the same table for one of those?  Perhaps 'en' really
does depend only on the UCA?





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 18:46  Jeff Davis <[email protected]>
  parent: Robert Haas <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 18:46 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, 2022-11-29 at 12:32 -0500, Robert Haas wrote:
> You know more about this than I do, for sure, so don't let my vote
> back the project into a bad spot.

I'm going back and forth myself. I haven't found a great answer here
yet.

>  But, yeah, the thing you mention
> here is what I'm worried about. Without a way to force a certain
> behavior for a certain particular collation, you don't have an escape
> valve if the global library ordering isn't doing what you want.

One bit of weirdness is that I may have found another ICU problem.
First, install 63.1, and you get (editing for clarity):

$ ls -l /path/to/libicui18n.so.63*
/path/to/libicui18n.so.63 -> libicui18n.so.63.1
/path/to/libicui18n.so.63.1

$ ls -l /path/to/libicuuc.so.63*
/path/to/libicuuc.so.63 -> libicuuc.so.63.1
/path/to/libicuuc.so.63.1

$ ls -l /path/to/libicudata.so.63*
/path/to/libicudata.so.63 -> libicudata.so.63.1
/path/to/lib/libicudata.so.63.1

$ ldd /path/to/libicui18n.so.63.1
        libicuuc.so.63 => /path/to/libicuuc.so.63
        libicudata.so.63 => /path/to/libicudata.so.63 

OK, now install 63.2. Then you get:

$ ls -l /path/to/libicui18n.so.63*
/path/to/libicui18n.so.63 -> libicui18n.so.63.2
/path/to/libicui18n.so.63.1
/path/to/libicui18n.so.63.2

$ ls -l /path/to/libicuuc.so.63*
/path/to/libicuuc.so.63 -> libicuuc.so.63.2
/path/to/libicuuc.so.63.1
/path/to/libicuuc.so.63.2

$ ls -l /path/to/libicudata.so.63*
/path/to/libicudata.so.63 -> libicudata.so.63.2
/path/to/libicudata.so.63.1
/path/to/libicudata.so.63.2

$ ldd /path/to/libicui18n.so.63.2
        libicuuc.so.63 => /path/to/libicuuc.so.63
        libicudata.so.63 => /path/to/libicudata.so.63

The problem is that the specific minor version 63.1 depends on only the
major version of its ICU link dependencies. When loading
libicui18n.so.63.1, you are actually pulling in libicuuc.so.63.2 and
libicudata.so.63.2.

When I tried this with Thomas's patch, it caused some confusing
problems. I inserted a check that, when you open a library, that the
requested and reported versions match, and the check failed when
multiple minors are installed. In other words, opening
libicui18n.so.63.1 reports a version of 63.2!

(Note: I compiled ICU with --enable-rpath, but I don't think it
matters.)

Summary: even locking down to a minor version does not seem to identify
a specific ICU library, because its shared library dependencies do not
reference a specific minor version.

> It's entirely possible that the scenario I'm worried about is too
> remote in practice to be concerned about. I don't know how this stuff
> works well enough to be certain. It's just that, on the basis of
> previous experience, (1) it's not that uncommon for people to
> actually
> end up in situations that we thought shouldn't ever happen and (2)
> code that deals with collations is more untrustworthy than average.

Yeah...


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 18:59  Jeff Davis <[email protected]>
  parent: Joe Conway <[email protected]>
  0 siblings, 2 replies; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 18:59 UTC (permalink / raw)
  To: Joe Conway <[email protected]>; Robert Haas <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, 2022-11-29 at 11:27 -0500, Joe Conway wrote:
> My vote is for something like #5. The collversion should indicate a 
> specific immutable ordering behavior.

Easier said than done:

https://www.postgresql.org/message-id/[email protected]

Even pointing at a specific minor version doesn't guarantee that
specific ICU code is loaded. It could also be a mix of different minor
versions that happen to be installed.

But if we ignore that problem for a moment, and assume that major
version is precise enough, let me make another proposal (not advocating
for this, but wanted to put it out there):

6. Create a new concept of a "locked down collation" that points at
some specific collation code (identified by some combination of library
version and collation version or whatever else can be used to identify
it). If a collation is locked down, it would never have a fallback or
any other magic, it would either find the code it's looking for, or
fail. If a collation is not locked down, it would look only in the
built-in ICU library, and warn if it detects some kind of change
(again, by whatever heuristic we think is reasonable).

#6 doesn't answer all of the problems I pointed out earlier:

https://www.postgresql.org/message-id/[email protected]

but could be a better starting place for answers.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:03  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 19:03 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; Jeremy Schneider <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-30 at 07:18 +1300, Thomas Munro wrote:
> On Wed, Nov 30, 2022 at 7:03 AM Jeremy Schneider
> <[email protected]> wrote:
> > It seems to me that the collator_version field is not a good
> > version
> > identifier to use.
> > 
> > Just taking a quick glance at the ICU home page right now, it shows
> > that
> > all of the last 5 versions of ICU have included "additions and
> > corrections" to locale data itself, including 68 to 69 where the
> > collator version did not change.
> > 
> > Is it possible that this "collator_version" only reflects the code
> > that
> > processes collation data to do comparisons/sorts, but it does not
> > reflect updates to the locale data itself?
> 
> I think it also includes the CLDR version for *some* locales.  From a
> quick look, that includes 'ar', 'ru', 'tr', 'zh'.  Jeff, would you
> mind sharing the same table for one of those?  Perhaps 'en' really
> does depend only on the UCA?

=# select * from pg_icu_collation_versions('ar') order by icu_version;
 icu_version | uca_version | collator_version                         
-------------+-------------+------------------                        
 50.2        | 6.2         | 58.0.0.50                                
 51.3        | 6.2         | 58.0.0.50                                
 52.2        | 6.2         | 58.0.0.50                                
 53.2        | 6.3         | 137.51.25
 54.2        | 7.0         | 137.56.26
 55.2        | 7.0         | 153.56.27.1
 56.2        | 8.0         | 153.64.28
 57.2        | 8.0         | 153.64.29
 58.3        | 9.0         | 153.72.30.3
 59.2        | 9.0         | 153.72.31.1
 60.3        | 10.0        | 153.80.32.1
 61.2        | 10.0        | 153.80.33
 62.2        | 11.0        | 153.88.33.8
 63.2        | 11.0        | 153.88.34
 64.2        | 12.1        | 153.97.35.8
 65.1        | 12.1        | 153.97.36
 66.1        | 13.0        | 153.14.36.8
 67.1        | 13.0        | 153.14.37
 68.2        | 13.0        | 153.14.38.8
 69.1        | 13.0        | 153.14.39
 70.1        | 14.0        | 153.112.40
(21 rows)


=# select * from pg_icu_collation_versions('zh') order by icu_version;
 icu_version | uca_version | collator_version 
-------------+-------------+------------------
 50.2        | 6.2         | 58.0.0.50
 51.3        | 6.2         | 58.0.0.50
 52.2        | 6.2         | 58.0.0.50
 53.2        | 6.3         | 137.51.25
 54.2        | 7.0         | 137.56.26
 55.2        | 7.0         | 153.56.27.1
 56.2        | 8.0         | 153.64.28
 57.2        | 8.0         | 153.64.29
 58.3        | 9.0         | 153.72.30.3
 59.2        | 9.0         | 153.72.31.1
 60.3        | 10.0        | 153.80.32.1
 61.2        | 10.0        | 153.80.33
 62.2        | 11.0        | 153.88.33.8
 63.2        | 11.0        | 153.88.34
 64.2        | 12.1        | 153.97.35.8
 65.1        | 12.1        | 153.97.36
 66.1        | 13.0        | 153.14.36.8
 67.1        | 13.0        | 153.14.37
 68.2        | 13.0        | 153.14.38.8
 69.1        | 13.0        | 153.14.39
 70.1        | 14.0        | 153.112.40
(21 rows)


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:34  Joe Conway <[email protected]>
  parent: Jeff Davis <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Joe Conway @ 2022-11-29 19:34 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; Robert Haas <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 11/29/22 13:59, Jeff Davis wrote:
> On Tue, 2022-11-29 at 11:27 -0500, Joe Conway wrote:
>> My vote is for something like #5. The collversion should indicate a 
>> specific immutable ordering behavior.
> 
> Easier said than done:
> https://www.postgresql.org/message-id/[email protected]
> 
> Even pointing at a specific minor version doesn't guarantee that
> specific ICU code is loaded. It could also be a mix of different minor
> versions that happen to be installed.

I understand that it is not easily done, but if the combination of 
collprovider + collversion does not represent specific immutable 
ordering behavior for a given locale, what value is there in tracking it?

-- 
Joe Conway
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:37  Jeff Davis <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 19:37 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, 2022-11-28 at 19:36 -0800, Jeff Davis wrote:
> On Mon, 2022-11-28 at 21:57 -0500, Robert Haas wrote:
> > That is ... astonishingly bad.
> 
> https://unicode-org.atlassian.net/browse/CLDR-16175

Oops, reported in CLDR instead of ICU. Moved to:

https://unicode-org.atlassian.net/browse/ICU-22215


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:38  Jeff Davis <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 19:38 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; +Cc: Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, 2022-11-29 at 10:46 -0800, Jeff Davis wrote:
> One bit of weirdness is that I may have found another ICU problem.

Reported as:

https://unicode-org.atlassian.net/browse/ICU-22216


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:41  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 19:41 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 8:03 AM Jeff Davis <[email protected]> wrote:
> On Wed, 2022-11-30 at 07:18 +1300, Thomas Munro wrote:
> > I think it also includes the CLDR version for *some* locales.  From a
> > quick look, that includes 'ar', 'ru', 'tr', 'zh'.  Jeff, would you
> > mind sharing the same table for one of those?  Perhaps 'en' really
> > does depend only on the UCA?
>
> =# select * from pg_icu_collation_versions('ar') order by icu_version;
>  icu_version | uca_version | collator_version
> -------------+-------------+------------------
>  50.2        | 6.2         | 58.0.0.50
>  51.3        | 6.2         | 58.0.0.50
>  52.2        | 6.2         | 58.0.0.50
>  53.2        | 6.3         | 137.51.25
>  54.2        | 7.0         | 137.56.26
>  55.2        | 7.0         | 153.56.27.1
>  56.2        | 8.0         | 153.64.28
>  57.2        | 8.0         | 153.64.29
>  58.3        | 9.0         | 153.72.30.3
>  59.2        | 9.0         | 153.72.31.1
>  60.3        | 10.0        | 153.80.32.1
>  61.2        | 10.0        | 153.80.33
>  62.2        | 11.0        | 153.88.33.8
>  63.2        | 11.0        | 153.88.34
>  64.2        | 12.1        | 153.97.35.8
>  65.1        | 12.1        | 153.97.36
>  66.1        | 13.0        | 153.14.36.8
>  67.1        | 13.0        | 153.14.37
>  68.2        | 13.0        | 153.14.38.8
>  69.1        | 13.0        | 153.14.39
>  70.1        | 14.0        | 153.112.40
> (21 rows)

Thanks.  So now we can see that the CLDR minor version is there too.
At a guess, in ICU 60 and before, it was the 4th component directly,
and from ICU 61 on, it's shifted left 3 bits.  I guess that means
those CLDR-dependent locales have higher frequency collversion
changes, including everyday "apt-get upgrade" (no major OS upgrade
required), assuming that Debian et al take those minor upgrades, while
others like 'en' should be stable for the whole ICU major version's
lifetime, and even across some ICU major version upgrades, because the
Unicode/UCA version changes more slowly.

Those CLDR-dependent locales therefore present us with a problem: as
discussed a while back, it's impossible to install two minor versions
of the same ICU major version with packages, and as Jeff has pointed
out in recent emails, even if you compile them yourself (which no one
really expects users to do), it doesn't really work because the
SONAMEs only have the major version, so the various libraries
that make up ICU will not be able to open each other correctly
(they'll follow symlinks to an arbitrary minor version).  (These two
things are not unrelated.)  So I probably need to remove the code that
claimed to support minor version addressing and go back to the
previous thinking that major will have to be enough.

In terms of user experience, I think that might mean that users of
'zh' who encounter warnings after a minor upgrade would therefore
really only have the options of REFRESHing and rebuilding, or
downgrading the package, because there's no way for us to access the
older version.  Users of 'en' probably only encounter collversion
changes when moving between OS releases with an ICU major version
change, and then the various schemes in this thread can help them
avoid the need to rebuild, until they eventually want to, if ever.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 19:52  Robert Haas <[email protected]>
  parent: Jeff Davis <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Robert Haas @ 2022-11-29 19:52 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Joe Conway <[email protected]>; Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 29, 2022 at 1:59 PM Jeff Davis <[email protected]> wrote:
> 6. Create a new concept of a "locked down collation" that points at
> some specific collation code (identified by some combination of library
> version and collation version or whatever else can be used to identify
> it). If a collation is locked down, it would never have a fallback or
> any other magic, it would either find the code it's looking for, or
> fail. If a collation is not locked down, it would look only in the
> built-in ICU library, and warn if it detects some kind of change
> (again, by whatever heuristic we think is reasonable).

It seems like it would be somewhat reasonable to allow varying levels
of specificity in saying which what suffix to append when calling
dlopen() on the ICU library. Like you could allow adding nothing,
which would find the system-default ICU, or you could add 53 to find
the default version of ICU 53, or you could 53.1 to pick a specific
minor version. The idea is that the symlinks in the filesystem would
be responsible for sorting out the meaning of the supplied string. The
way that minor versions work may preclude having this work as well as
one might hope, though.

I continue to be confused about why collation maintainers think that
it's OK to whack stuff around in minor versions. The thought that
people might use collations to sort data that needs to stay sorted
after upgrading the library seems to be an alien one, and it doesn't
really seem like libicu is a whole lot better than libc, either.

-- 
Robert Haas
EDB: http://www.enterprisedb.com





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 20:00  Thomas Munro <[email protected]>
  parent: Robert Haas <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 20:00 UTC (permalink / raw)
  To: Robert Haas <[email protected]>; +Cc: Jeff Davis <[email protected]>; Joe Conway <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 8:52 AM Robert Haas <[email protected]> wrote:
> On Tue, Nov 29, 2022 at 1:59 PM Jeff Davis <[email protected]> wrote:
> > 6. Create a new concept of a "locked down collation" that points at
> > some specific collation code (identified by some combination of library
> > version and collation version or whatever else can be used to identify
> > it). If a collation is locked down, it would never have a fallback or
> > any other magic, it would either find the code it's looking for, or
> > fail. If a collation is not locked down, it would look only in the
> > built-in ICU library, and warn if it detects some kind of change
> > (again, by whatever heuristic we think is reasonable).
>
> It seems like it would be somewhat reasonable to allow varying levels
> of specificity in saying which what suffix to append when calling
> dlopen() on the ICU library. Like you could allow adding nothing,
> which would find the system-default ICU, or you could add 53 to find
> the default version of ICU 53, or you could 53.1 to pick a specific
> minor version. The idea is that the symlinks in the filesystem would
> be responsible for sorting out the meaning of the supplied string. The
> way that minor versions work may preclude having this work as well as
> one might hope, though.

I'm struggling to understand what's new about proposal #6.  The
earlier proposals except #1 already contemplated different levels of
locked-down-ness.  For example in the libversion-as-provider idea, we
said you could use just provider = ICU (warn me if the collverison
changes, but always use the "default" library and carry on, pretty
much like today except perhaps "the default" can be changed with a
GUC), or you could be more specific and say provider = ICU63.  (We
also mentioned ICU63_2 as a third level of specificity, but maybe
that's practically impossible.)  And it was the same for the other
ideas, just encoded in different ways.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 20:21  Jeff Davis <[email protected]>
  parent: Joe Conway <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 20:21 UTC (permalink / raw)
  To: Joe Conway <[email protected]>; Robert Haas <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, 2022-11-29 at 14:34 -0500, Joe Conway wrote:
> I understand that it is not easily done, but if the combination of 
> collprovider + collversion does not represent specific immutable 
> ordering behavior for a given locale

Given the u_versionToString() bug, we know the version string could end
up being the same between two different collation versions (e.g.
153.104 and 153.14). So that really undermines the credibility of ICU's
collation versions (at least the strings, which is what we store in
collversion).

But if we ignore that bug, do we have evidence that the actual versions
could be the same for collations that sort differently? It's worth
exploring, to be sure, but right now I don't know of a case.

> , what value is there in tracking [collation version]?

Similarly, what is the value in tracking the library minor versions, if
when you open libicui18n.63.1, you may end up with a mix of code
between 63.1 and 63.2?

That doesn't mean it's impossible. We could attach collations to a
library major version, and tell administrators that once they install a
major version in icu_library_path, they never touch that major version
again (no updates or new minors, only new majors). #6 might be a good
approach to facilitate this best practice. We'd then probably need to
change collversion to be a library major version, and then come up with
a migration path from 15 -> 16. Or we could store both library major
version and collversion, and verify both.

-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 20:59  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 20:59 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-30 at 08:41 +1300, Thomas Munro wrote:
> In terms of user experience, I think that might mean that users of
> 'zh' who encounter warnings after a minor upgrade would therefore
> really only have the options of REFRESHing and rebuilding, or
> downgrading the package, because there's no way for us to access the
> older version.  Users of 'en' probably only encounter collversion
> changes when moving between OS releases with an ICU major version
> change, and then the various schemes in this thread can help them
> avoid the need to rebuild, until they eventually want to, if ever.

I installed the first minor release for each major, and got some new
tables. I think we can all agree that it's a lot easier to work with
information once it's in table form.

Here's what I found for the 'ar' locale (firstminor/lastminor are the
icu library versions, firstcollversion/lastcollversion are their
respective collation versions for the given locale):

 firstminor | lastminor | firstcollversion | lastcollversion 
------------+-----------+------------------+-----------------
 60.1       | 60.3      | 153.80.32        | 153.80.32.1
 64.1       | 64.2      | 153.96.35        | 153.97.35.8
 68.1       | 68.2      | 153.14.38        | 153.14.38.8
(3 rows)

For 'en':

 firstminor | lastminor | firstcollversion | lastcollversion 
------------+-----------+------------------+-----------------
 64.1       | 64.2      | 153.96           | 153.97
(1 row)

And for 'zh':

 firstminor | lastminor | firstcollversion | lastcollversion 
------------+-----------+------------------+-----------------
 60.1       | 60.3      | 153.80.32        | 153.80.32.1
 64.1       | 64.2      | 153.96.35        | 153.97.35.8
 68.1       | 68.2      | 153.14.38        | 153.14.38.8
(3 rows)

It looks like collation versions do change in minor releases. It looks
like it's *not* safe to lock a collation to a major version *if* that
major version could be updated to a new minor. And we can't lock to a
minor, as I said earlier. Therefore, once we lock a collation down to a
major release, we better keep that in the icu_library_path, and never
touch it, and never install a new minor for that major.

Then again, maybe some of these are just about how the version is
reported... maybe 153.80.32 and 153.80.32.1 are really the same
version? But 64.1 -> 64.2 looks like a real difference.

I suppose the next step is to test with actual data and find
differences?


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 21:29  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 21:29 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 9:59 AM Jeff Davis <[email protected]> wrote:
> Here's what I found for the 'ar' locale (firstminor/lastminor are the
> icu library versions, firstcollversion/lastcollversion are their
> respective collation versions for the given locale):
>
>  firstminor | lastminor | firstcollversion | lastcollversion
> ------------+-----------+------------------+-----------------
>  60.1       | 60.3      | 153.80.32        | 153.80.32.1
>  64.1       | 64.2      | 153.96.35        | 153.97.35.8
>  68.1       | 68.2      | 153.14.38        | 153.14.38.8
> (3 rows)

Right, this fits with what I said earlier: the third component is CLDR
major, fourth component is CLDR minor except from ICU 61 on the CLDR
minor is << 3'd (X.X.38.8 means CLDR 38.1).  I wrote something about
that particular CLDR upgrade that happened in ICU 68 back here, with a
link to the CLDR change list:

https://www.postgresql.org/message-id/[email protected]...

TL;DR that particular CLDR change didn't actually affect collations,
it affected other locale stuff we don't care about (timezones etc).
We probably have to assume that any CLDR change *might* affect us,
though, unless we can find a written policy somewhere that says CLDR
minor changes never change sort order.  But I wouldn't want to get
into 2nd guessing their ucol_getVersion() format, and if they knew
that minor changes didn't affect sort order they presumably wouldn't
have included it in the recipe, so I think we simply have to treat it
as opaque and assume that ucol_getVersion() change means what it says
on the tin: sort order might have changed.

> I suppose the next step is to test with actual data and find
> differences?

Easier to read the published CLDR deltas, but I'm not sure it'd tell
us much about what *could* happen in future releases...





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 21:41  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Jeff Davis @ 2022-11-29 21:41 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; Robert Haas <[email protected]>; +Cc: Joe Conway <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-30 at 09:00 +1300, Thomas Munro wrote:
> I'm struggling to understand what's new about proposal #6.

Perhaps it's just a slight variant; I'm not sure. It's not a complete
proposal yet.

The difference I had in mind is that it would treat the built-in ICU
differently from what is found in icu_library_path. I think that could
remove confusion over what happens when you upgrade the system's ICU
library.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-29 21:52  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-29 21:52 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Robert Haas <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 8:38 AM Jeff Davis <[email protected]> wrote:
> On Tue, 2022-11-29 at 10:46 -0800, Jeff Davis wrote:
> > One bit of weirdness is that I may have found another ICU problem.
>
> Reported as:
>
> https://unicode-org.atlassian.net/browse/ICU-22216

I'm no expert on loader/linker arcana but I have a feeling this is a
dead end.  It's an ancient Unix or at least elf-era Unix convention
that SONAMEs have major versions only, because major versions are the
basis of ABI stability.

As a workaround with an already built ICU, I think you could use elf
editing tools like "patchelf" to change the SONAME and DT_NEEDED to
include the minor version.  Or you could convince the build/link
scripts to set them that way in the first place, but no distro would
want to do that as it would cause lots of executables to fail to load
when the next ICU minor comes out.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-30 00:25  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-30 00:25 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Robert Haas <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-30 at 10:52 +1300, Thomas Munro wrote:
> On Wed, Nov 30, 2022 at 8:38 AM Jeff Davis <[email protected]> wrote:
> > On Tue, 2022-11-29 at 10:46 -0800, Jeff Davis wrote:
> > > One bit of weirdness is that I may have found another ICU
> > > problem.
> > 
> > Reported as:
> > 
> > https://unicode-org.atlassian.net/browse/ICU-22216
> 
> I'm no expert on loader/linker arcana but I have a feeling this is a
> dead end.  It's an ancient Unix or at least elf-era Unix convention
> that SONAMEs have major versions only, because major versions are the
> basis of ABI stability.

It's possible that it's more a problem of how they are doing it: the
specific version is coming from a dependency rather than the library
itself. The results are surprising, so I figured it's worth a report.
Let's see what they say.

Regardless, even if they did make a change, it's not going to help us
anytime soon. We can't rely on any scheme that involves multiple minor
versions for a single major version being installed at once. That means
that, if you create a collation depending on ICU X.Y, and then it gets
upgraded to X.(Y+1), and you create another collation depending on that
library version, you are stuck.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-30 00:32  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-11-30 00:32 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, 2022-11-30 at 10:29 +1300, Thomas Munro wrote:
> On Wed, Nov 30, 2022 at 9:59 AM Jeff Davis <[email protected]> wrote:
> > Here's what I found for the 'ar' locale (firstminor/lastminor are
> > the
> > icu library versions, firstcollversion/lastcollversion are their
> > respective collation versions for the given locale):
> > 
> >  firstminor | lastminor | firstcollversion | lastcollversion
> > ------------+-----------+------------------+-----------------
> >  60.1       | 60.3      | 153.80.32        | 153.80.32.1
> >  64.1       | 64.2      | 153.96.35        | 153.97.35.8
> >  68.1       | 68.2      | 153.14.38        | 153.14.38.8
> > (3 rows)
> 
> Right, this fits with what I said earlier: the third component is
> CLDR
> major, fourth component is CLDR minor except from ICU 61 on the CLDR
> minor is << 3'd (X.X.38.8 means CLDR 38.1).

What about 64.1 -> 64.2? That changed the *second* component from 96 ->
97. Are we agreed that collations can materially change in minor ICU
releases?


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-30 00:50  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-11-30 00:50 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 1:32 PM Jeff Davis <[email protected]> wrote:
> On Wed, 2022-11-30 at 10:29 +1300, Thomas Munro wrote:
> > On Wed, Nov 30, 2022 at 9:59 AM Jeff Davis <[email protected]> wrote:
> > > Here's what I found for the 'ar' locale (firstminor/lastminor are
> > > the
> > > icu library versions, firstcollversion/lastcollversion are their
> > > respective collation versions for the given locale):
> > >
> > >  firstminor | lastminor | firstcollversion | lastcollversion
> > > ------------+-----------+------------------+-----------------
> > >  60.1       | 60.3      | 153.80.32        | 153.80.32.1
> > >  64.1       | 64.2      | 153.96.35        | 153.97.35.8
> > >  68.1       | 68.2      | 153.14.38        | 153.14.38.8
> > > (3 rows)
> >
> > Right, this fits with what I said earlier: the third component is
> > CLDR
> > major, fourth component is CLDR minor except from ICU 61 on the CLDR
> > minor is << 3'd (X.X.38.8 means CLDR 38.1).
>
> What about 64.1 -> 64.2? That changed the *second* component from 96 ->
> 97. Are we agreed that collations can materially change in minor ICU
> releases?

That means that the Unicode/UCA version switched from 12 to 12.1, so
that's a confirmed sighting of a UCA minor version bump within one ICU
major version.  Let's see what the purpose of that Unicode minor
release was[1]:

"Unicode 12.1 adds exactly one character, for a total of 137,929 characters.

The new character added to Version 12.1 is:

U+32FF SQUARE ERA NAME REIWA

Version 12.1 adds that single character to enable software to be
rapidly updated to support the new Japanese era name in calendrical
systems and date formatting. The new Japanese era name was officially
announced on April 1, 2019, and is effective as of May 1, 2019."

Wow!

Wikipedia says[2] "the "rei" character 令 has never appeared before".

The sort order of characters that didn't previously exist is a special
topic.  In theory they can't hurt you because you shouldn't have been
using them, but PostgreSQL doesn't enforce that (other systems do), so
you could be exposed to a change from whatever default ordering the
non-existent codepoint had for random implementation reasons to some
deliberate ordering which may or may not be the same.

Are all Unicode/UCA minor versions of that type?  I dunno.  Something
to research, but [3] is far too vague and [4] is about other problems.

[1] https://unicode.org/versions/Unicode12.1.0/
[2] https://en.wikipedia.org/wiki/Reiwa
[3] https://www.unicode.org/versions/#major_minor
[4] https://www.unicode.org/policies/stability_policy.html





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-30 00:54  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Thomas Munro @ 2022-11-30 00:54 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Robert Haas <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Finnerty, Jim <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 1:25 PM Jeff Davis <[email protected]> wrote:
> On Wed, 2022-11-30 at 10:52 +1300, Thomas Munro wrote:
> > On Wed, Nov 30, 2022 at 8:38 AM Jeff Davis <[email protected]> wrote:
> > > On Tue, 2022-11-29 at 10:46 -0800, Jeff Davis wrote:
> > > https://unicode-org.atlassian.net/browse/ICU-22216
> >
> > I'm no expert on loader/linker arcana but I have a feeling this is a
> > dead end.  It's an ancient Unix or at least elf-era Unix convention
> > that SONAMEs have major versions only, because major versions are the
> > basis of ABI stability.
>
> It's possible that it's more a problem of how they are doing it: the
> specific version is coming from a dependency rather than the library
> itself. The results are surprising, so I figured it's worth a report.
> Let's see what they say.
>
> Regardless, even if they did make a change, it's not going to help us
> anytime soon. We can't rely on any scheme that involves multiple minor
> versions for a single major version being installed at once. That means
> that, if you create a collation depending on ICU X.Y, and then it gets
> upgraded to X.(Y+1), and you create another collation depending on that
> library version, you are stuck.

Mainstream package maintainers aren't going to let that happen anyway
as discussed, so this would always be a fairly specialised concern.
Maybe someone in our community would be motivated to publish a repo
full of mutant packages that don't conflict with each other and that
have specially modified DT_NEEDED, or are rolled into one single
library so the DT_NEEDED problem goes away.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-11-30 01:00  Michael Paquier <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Michael Paquier @ 2022-11-30 01:00 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeff Davis <[email protected]>; Jeremy Schneider <[email protected]>; Peter Eisentraut <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Wed, Nov 30, 2022 at 01:50:51PM +1300, Thomas Munro wrote:
> The new character added to Version 12.1 is:
> 
> U+32FF SQUARE ERA NAME REIWA
> 
> Version 12.1 adds that single character to enable software to be
> rapidly updated to support the new Japanese era name in calendrical
> systems and date formatting. The new Japanese era name was officially
> announced on April 1, 2019, and is effective as of May 1, 2019."
> 
> Wow!

Wow++.  I didn't know this one.

> Wikipedia says[2] "the "rei" character 令 has never appeared before".

At least there was some time ahead to prepare for the switch from "平
成" to "令和".  Things were much "funnier" when the era has switched
from "昭和" to "平成", as the sudden death of the emperor has required
Japan to switch to a new calendar very suddenly back in the day..
I've heard this was quite a mess for folks in IT back then, especially
for public agencies.
--
Michael


Attachments:

  [application/pgp-signature] signature.asc (833B, ../../[email protected]/2-signature.asc)
  download

^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-01 13:22  Dagfinn Ilmari Mannsåker <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Dagfinn Ilmari Mannsåker @ 2022-12-01 13:22 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Robert Haas <[email protected]>; Thomas Munro <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

Jeff Davis <[email protected]> writes:

> On Mon, 2022-11-28 at 19:36 -0800, Jeff Davis wrote:
>> On Mon, 2022-11-28 at 21:57 -0500, Robert Haas wrote:
>> > That is ... astonishingly bad.
>> 
>> https://unicode-org.atlassian.net/browse/CLDR-16175
>
> Oops, reported in CLDR instead of ICU. Moved to:
>
> https://unicode-org.atlassian.net/browse/ICU-22215

Out of morbid curiosity I went source diving, and the culprit is this
bit (which will also break if a version component ever goes above 999):

    /* write the decimal field value */
    field=versionArray[part];
    if(field>=100) {
        *versionString++=(char)('0'+field/100);
        field%=100;
    }
    if(field>=10) {
        *versionString++=(char)('0'+field/10);
        field%=10;
    }
    *versionString++=(char)('0'+field);

(https://sources.debian.org/src/icu/72.1-3/source/common/putil.cpp#L2308)

because apparently snprintf() is too hard?

- ilmari





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-05 03:12  Thomas Munro <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 2 replies; 57+ messages in thread

From: Thomas Munro @ 2022-12-05 03:12 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Nov 29, 2022 at 7:51 PM Jeff Davis <[email protected]> wrote:
> On Sat, 2022-11-26 at 18:27 +1300, Thomas Munro wrote:
> > On Thu, Nov 24, 2022 at 5:48 PM Thomas Munro <[email protected]>
> > wrote:
> > > On Thu, Nov 24, 2022 at 3:07 PM Jeff Davis <[email protected]>
> > > wrote:
> > > > I'd vote for 1 on the grounds that it's easier to document and
> > > > understand a single collation version, which comes straight from
> > > > ucol_getVersion(). This approach makes it a separate problem to
> > > > find
> > > > the collation version among whatever libraries the admin can
> > > > provide;
> > > > but adding some observability into the search should mitigate any
> > > > confusion.
> > >
> > > OK, it sounds like I should code that up next.
> >
> > Here's the first iteration.
>
> Thank you.

Thanks for the review.  Responses further down.  And thanks also for
the really interesting discussion about how the version numbers work
(or in some cases, don't work...), and practical packaging and linking
problems.

To have a hope of making something happen for PG16, which I think
means we need a serious contender patch in the next few weeks, we
really need to make some decisions.  I enjoyed trying out
search-by-collversion, but it's still not my favourite.  On the ballot
we have two main questions:

1.  Should we commit to search-by-collversion, or one of the explicit
library version ideas, and if the latter, which?
2.  Should we try to support being specific about minor versions (in
various different ways according to the choice made for #1)?

My tentative votes are:

1.  I think we should seriously consider provider = ICU63.  I still
think search-by-collversion is a little too magical, even though it
clearly can be made to work.  Of the non-magical systems, I think
encoding the choice of library into the provider name would avoid the
need to add a second confusing "X_version" concept alongside our
existing "X_version" columns in catalogues and DDL syntax, while still
making it super clear what is going on.  This would include adding DDL
commands so you can do ALTER DATABASE/COLLATION ... PROVIDER = ICU63
to make warnings go way.

2.  I think we should ignore minor versions for now (other than
reporting them in the relevant introspection functions), but not make
any choices that would prevent us from changing our mind about that in
a later release.  For example, having two levels of specificity ICU
and ICU68  in the libver-in-provider-name design wouldn't preclude us
from adding support for ICU68_2 later

I haven't actually tried that design out in code yet, but I'm willing
to try to code that up very soon.  So no new patch from me yet.  Does
anyone else want to express a view?

> Proposed changes:
>
> * I attached a first pass of some documentation.

Thanks.  Looks pretty good, and much of it would stay if we changed to
one of the other models.

> * Should be another GUC to turn WARNING into an ERROR. Useful at least
> for testing; perhaps too dangerous for production.

OK, will add that into the next version.

> * The libraries should be loaded in a more diliberate order. The "*"
> should be expanded in a descending fashion so that later versions are
> preferred.

Yeah, I agree.

> * GUCs should be validated.

Will do.

> * Should validate that loaded library has expected version.

Will do.

> * We need to revise or remove pg_collation_actual_version() and
> pg_database_collation_actual_version().

I never liked that use of the word "actual"...

> * The GUCs are PGC_SUSET, but don't take effect because
> icu_library_list_fully_loaded is never reset.

True.  Just rought edges because I was trying to prototype
search-by-collversion fast.  Will consider this for the next version.

> * The extra collations you're adding at bootstrap time are named based
> on the library major version. I suppose it might be more "proper" to
> name them based on the collation version, but that would be more
> verbose, so I won't advocate for that. Just pointing it out.

Ah, yes, the ones with names like "en-US-x-icu68".  I agree that made
a little less sense in the search-by-collversion patch.  Maybe we
wouldn't want these at all in the search-by-collversion model.  But I
think they're perfect the way they are in the provider = ICU68 model.
The other idea I considered ages ago was that we could use namespaces:
you could "icu68.en-US", or just "en-US" in some contexts to get what
your search path sees, but that all seemed a little too cute and not
really like anything else we do with system-created catalogues, so I
gave that idea up.

> * It looks hard (or impossible) to mix multiple ICU libraries with the
> same major version and different minor versions. That's because,
> e.g., libicui18n.so.63.1 links against libicuuc.63 and libicudata.63,
> and when you install ICU 63.2, those dependencies get clobbered with
> the 63.2 versions. That fails the sanity check I proposed above about
> the library version number matching the requested library version
> number. And it also just seems wrong -- why would you have minor-
> version precision about an ICU library but then only major-version
> precision about the ICU dependencies of that library? Doesn't that
> defeat the whole purpose of this naming scheme? (Maybe another ICU
> bug?).

I don't think it's a bug exactly.  That scheme is designed to
advertise ABI stability, and not intended to support parallel
installation of minor versions.  It does seem a little silly for
libraries that are shipped together as one atomic unit not to use
fully qualified dependency names, though.

I think there would be various technical solutions, if you're prepared
to give up existing ready-made packages and build stuff yourself.
Install them into different directories with different DT_RPATH so
they can't see each other (but then our icu_library_path needs to
support a list of paths or it won't find these ones which will have to
be not in the usual system path), or clobber the DT_NEEDED (but I
guess not the DT_SONAME) to mention the minor version, and equivalent
concepts for other non-elf systems (at a glance the same problem
applies on macOS), or re-roll the libraries into a single .so.  Or
convince them to support a single library build mode (maybe there is
one already?  I couldn't find it).

That's all a bit against the grain for now, and makes me want to
abandon the notion of minor versions completely for now but leave the
option open for later exploration.

In the meantime, I think the feature is still pretty useful.  For
example, it helps you with the common case of a major OS upgrade or
streaming replication across major OS versions: just find the right
.deb/rpm/whatever for the older one, and install it, until you're
ready to upgrade and REFRESH.  The story is not quite as good for
someone with an index full of Chinese or Turkish text who gets a
surprise warning after a minor apt-get update, because the Japanese
have decided to invent a new character.  We can't offer a nice
solution to that: they have to determine that it is safe to REFRESH to
clear the warning, with or without rebuild, or downgrade/pin the ICU
package until they are ready to REFRESH.  But that is already the case
today and this patch neither helps nor hinders.  The only reason we
didn't know about this pre-existing type of problem is because
(approximately) nobody uses ICU yet, because it wasn't available as a
database default yet.

> Minor comments:
>
> * ICU_I18N is defined in make_icu_library_name() but used outside of
> it. One solution might be to have it return both library names to the
> caller and rename it as make_icu_library_names().

Good idea, will do.

> * get_icu_function() could use a clarifying comment or a better name.
> Something that communicates that you are looking for the function in
> the given library with the given major version number (which may or may
> not be needed depending on how the library was compiled).

Agreed.

> * typo in comment over make_icu_collator:
> s/u_getVersion/ucol_getVersion/

Thanks.

> * The return value of make_icu_collator() seems backwards to me,
> stylistically. I typically see the false-is-good pattern with integer
> returns.

Agreed.

> * weird bracketing style in get_icu_collator for the "else"

Yep.

> >   The version rosetta stone functions look like this:
> >
> > postgres=# select * from pg_icu_library_versions();
> >  icu_version | unicode_version | cldr_version
> > -------------+-----------------+--------------
> >  67.1        | 13.0            | 37.0
> >  63.1        | 11.0            | 34.0
> >  57.1        | 8.0             | 29.0
> > (3 rows)
> >
> > postgres=# select * from pg_icu_collation_versions('zh');
> >  icu_version | uca_version | collator_version
> > -------------+-------------+------------------
> >  67.1        | 13.0        | 153.14.37
> >  63.1        | 11.0        | 153.88.34
> >  57.1        | 8.0         | 153.64.29
> > (3 rows)
>
> I like these functions.

Yeah, they've been quite educational.  Now I'm wondering what form
these functions would take in the provider = ICU68 patch.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-05 15:45  Robert Haas <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 0 replies; 57+ messages in thread

From: Robert Haas @ 2022-12-05 15:45 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Jeff Davis <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Sun, Dec 4, 2022 at 10:12 PM Thomas Munro <[email protected]> wrote:
> My tentative votes are:
>
> 1.  I think we should seriously consider provider = ICU63.  I still
> think search-by-collversion is a little too magical, even though it
> clearly can be made to work.  Of the non-magical systems, I think
> encoding the choice of library into the provider name would avoid the
> need to add a second confusing "X_version" concept alongside our
> existing "X_version" columns in catalogues and DDL syntax, while still
> making it super clear what is going on.  This would include adding DDL
> commands so you can do ALTER DATABASE/COLLATION ... PROVIDER = ICU63
> to make warnings go way.

+1. I wouldn't lose any sleep if we picked a different non-magical
option, but I think this is probably my favorite of the
explicit-library-version options (though it is close) and I like it
better than search-by-collversion.

(It's possible that I'm wrong to like it better, but I do.)

> 2.  I think we should ignore minor versions for now (other than
> reporting them in the relevant introspection functions), but not make
> any choices that would prevent us from changing our mind about that in
> a later release.  For example, having two levels of specificity ICU
> and ICU68  in the libver-in-provider-name design wouldn't preclude us
> from adding support for ICU68_2 later

+1.

-- 
Robert Haas
EDB: http://www.enterprisedb.com





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-05 17:41  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  1 sibling, 1 reply; 57+ messages in thread

From: Jeff Davis @ 2022-12-05 17:41 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Mon, 2022-12-05 at 16:12 +1300, Thomas Munro wrote:
> 1.  I think we should seriously consider provider = ICU63.  I still
> think search-by-collversion is a little too magical, even though it
> clearly can be made to work.  Of the non-magical systems, I think
> encoding the choice of library into the provider name would avoid the
> need to add a second confusing "X_version" concept alongside our
> existing "X_version" columns in catalogues and DDL syntax, while
> still
> making it super clear what is going on.

As I understand it, this is #2 in your previous list?

Can we put the naming of the provider into the hands of the user, e.g.:

  CREATE COLLATION PROVIDER icu63 TYPE icu
    AS '/path/to/libicui18n.so.63', '/path/to/libicuuc.so.63';

In this model, icu would be a "provider kind" and icu63 would be the
specific provider, which is named by the user.

That seems like the least magical approach, to me. We need an ICU
library; the administrator gives us one that looks like ICU; and we're
happy.

It avoids a lot of the annoyances we're discussing, and puts the power
in the hands of the admin. If they want to allow minor version updates,
they specify the library with .so.63, and let the symlinking handle it.

Of course, we can still do some sanity checks (WARNINGs or ERRORs) when
we think something is going wrong; like the version of ICU is too new,
or the reported version (ucol_getVersion()) doesn't match what's in
collversion. But we basically get out of the business of understanding
ICU versioning and leave that up to the administrator.

It's easier to document, and would require fewer GUCs (if any). And it
avoids mixing version information from another project into our data
model.


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS







^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-05 17:45  Joe Conway <[email protected]>
  parent: Jeff Davis <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Joe Conway @ 2022-12-05 17:45 UTC (permalink / raw)
  To: Jeff Davis <[email protected]>; Thomas Munro <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On 12/5/22 12:41, Jeff Davis wrote:
> On Mon, 2022-12-05 at 16:12 +1300, Thomas Munro wrote:
>> 1.  I think we should seriously consider provider = ICU63.  I still
>> think search-by-collversion is a little too magical, even though it
>> clearly can be made to work.  Of the non-magical systems, I think
>> encoding the choice of library into the provider name would avoid the
>> need to add a second confusing "X_version" concept alongside our
>> existing "X_version" columns in catalogues and DDL syntax, while
>> still
>> making it super clear what is going on.
> 
> As I understand it, this is #2 in your previous list?
> 
> Can we put the naming of the provider into the hands of the user, e.g.:
> 
>    CREATE COLLATION PROVIDER icu63 TYPE icu
>      AS '/path/to/libicui18n.so.63', '/path/to/libicuuc.so.63';
> 
> In this model, icu would be a "provider kind" and icu63 would be the
> specific provider, which is named by the user.
> 
> That seems like the least magical approach, to me. We need an ICU
> library; the administrator gives us one that looks like ICU; and we're
> happy.

+1

I like this. The provider kind defines which path we take in our code, 
and the specific library unambiguously defines a specific collation 
behavior (I think, ignoring bugs?)

-- 
Joe Conway
PostgreSQL Contributors Team
RDS Open Source Databases
Amazon Web Services: https://aws.amazon.com






^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-05 21:33  Thomas Munro <[email protected]>
  parent: Joe Conway <[email protected]>
  0 siblings, 1 reply; 57+ messages in thread

From: Thomas Munro @ 2022-12-05 21:33 UTC (permalink / raw)
  To: Joe Conway <[email protected]>; +Cc: Jeff Davis <[email protected]>; Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, Dec 6, 2022 at 6:45 AM Joe Conway <[email protected]> wrote:
> On 12/5/22 12:41, Jeff Davis wrote:
> > On Mon, 2022-12-05 at 16:12 +1300, Thomas Munro wrote:
> >> 1.  I think we should seriously consider provider = ICU63.  I still
> >> think search-by-collversion is a little too magical, even though it
> >> clearly can be made to work.  Of the non-magical systems, I think
> >> encoding the choice of library into the provider name would avoid the
> >> need to add a second confusing "X_version" concept alongside our
> >> existing "X_version" columns in catalogues and DDL syntax, while
> >> still
> >> making it super clear what is going on.
> >
> > As I understand it, this is #2 in your previous list?
> >
> > Can we put the naming of the provider into the hands of the user, e.g.:
> >
> >    CREATE COLLATION PROVIDER icu63 TYPE icu
> >      AS '/path/to/libicui18n.so.63', '/path/to/libicuuc.so.63';
> >
> > In this model, icu would be a "provider kind" and icu63 would be the
> > specific provider, which is named by the user.
> >
> > That seems like the least magical approach, to me. We need an ICU
> > library; the administrator gives us one that looks like ICU; and we're
> > happy.
>
> +1
>
> I like this. The provider kind defines which path we take in our code,
> and the specific library unambiguously defines a specific collation
> behavior (I think, ignoring bugs?)

OK, I'm going to see what happens if I try to wrangle that stuff into
a new catalogue table.





^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* Re: Collation version tracking for macOS
@ 2022-12-08 05:56  Jeff Davis <[email protected]>
  parent: Thomas Munro <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Jeff Davis @ 2022-12-08 05:56 UTC (permalink / raw)
  To: Thomas Munro <[email protected]>; Joe Conway <[email protected]>; +Cc: Peter Eisentraut <[email protected]>; Jeremy Schneider <[email protected]>; Peter Geoghegan <[email protected]>; Nasby, Jim <[email protected]>; Tom Lane <[email protected]>; pgsql-hackers

On Tue, 2022-12-06 at 10:33 +1300, Thomas Munro wrote:
> OK, I'm going to see what happens if I try to wrangle that stuff into
> a new catalogue table.

I've been hacking on a major refactor of the locale-related code. I
attached my progress and I think several patches are ready.

The main motivation is that I was frustrated by the special cases
everywhere. I wanted to make it easier to hack on this code going
forward, now that we are adding even more complexity for multiple ICU
libraries.

I'm posting to this thread (rather than my previous refactoring
thread[1]) because a lot of the people interested in working on this
code are here. So, if you like (or don't like) the structure of these
changes, please let me know.

Changes:
  * Introduce pg_locale_internal.h to hide all USE_ICU code,
    including all callers of the ICU routines. The files that
    still need to include pg_locale_internal.h are:
    - pg_locale.c
    - regc_pg_locale.c
    - formatting.c
    - like.c
    - like_support.c
    - collationcmds.c
  * Other callers (in files that don't include 
    pg_locale_internal.h) don't need to branch based on the
    provider, platform, database encoding, USE_ICU,
    HAVE_LOCALE_T, etc.
  * ICU and libc are treated the same way in more places.
  * I made it so pg_locale_t is constructed first, then
    moved to TopMemoryContext, so that it won't leak in
    TopMemoryContext if errors are encountered.
  * Introduce pg_strcoll, pg_strncoll, pg_strxfrm, and pg_strnxfrm
    so that varlena/hash/verchar code doesn't worry about the
    details.
  * Add method structure pg_icu_library, borrowed from Thomas's
    patch, that provides one convenient place to provide
    multiple-ICU-library support.
  * Add a hook that allows you to fill in the pg_icu_library
    structure however you want while a pg_locale_t is being
    constructed. This allows do-it-yourself ICU library
    lockdown.

On the negative side, it increases the line count. Part of that is
because adding indirection for the ICU library is just more lines of
code, but a lot of it is just that I used a lot of smaller functions.
Perhaps my style is a bit verbose?

Even though we're close to consensus on how we should offer control
over the ICU libraries, having the hook may be useful for
experimentation, testing, or as a last resort. Right now the hook has
limited information to use to find the right library -- just the ICU
collation name and the version, because that's what we have in the
catalog. But I assume the patch Thomas is working on will change that.

Performance:

I did brief performance sanity tests on several paths and the results
are unremarkable (which is generally good for a refactor). On the path
I was watching most closely, ICU/UTF8/en-US-x-icu, it came in about 2%
faster, which was a pleasant surprise. This was true both when I
disabled abbreviated keys (to stress localized comparison paths) and
also with abbreviated keys enabled. My previous refactoring work[1]
ended up a percent or two slower. My guess right now is that I moved
some code around after I noticed that ICU accepts NUL-terminated
strings (by specifying the lenght as -1), and that helped. But I'll
need to profile and look more closely to be more certain of my results,
these are preliminary.

There are a few things that could be done differently:

  * I am still a bit confused about why someone would want a collation
with a different lc_collate and lc_ctype in libc; and assuming there is
a reason, why it can't be done with ICU. The way I did the refactoring
tries to accommodate them as different concepts, but I can rip that
out.

  * In theory, we could also support multilib libc, and an associated
get_libc_library() and hook, but there are a couple big challenges.
Firstly, if it's the default locale, it relies on setlocale(), so we'd
have to figure out what to do about that. Second, having an a second
version of glibc on your system is not as normal or trivial as having a
second version of ICU.

  * I made the hook simple, but all it can do is replace the ICU
library. It's possible that it would want to construct it's own entire
pg_locale_t for some reason, and keep more complex state in a private
pointer, or something like that. It seemed better to keep it simple,
but maybe someone would want more flexibility there?

  * I used the library indirection for pretty much all ICU calls,
including the ucnv_ and the uloc_ functions. I did this mainly because,
if we are so paranoid about ICU changing in subtle ways, we might as
well make it possible to lock down everything. I can rip this out, too,
but it didn't add many lines.

Loose ends:

  * I need to do something with get_collation_actual_version. I had an
earlier iteration that went through pg_newlocale() and then queried the
resulting pg_locale_t structure, but that changed the error paths in a
way that failed a couple tests, so I left that out.

  * Error paths could be improved further to make sure that libc
locale_t and UCollator structures are freed in error paths during
construction. I was thinking about using resowner for this, and then if
the pg_locale_t structure gets moved to TopMemoryContext, just doing a
ResourceOwnerForget. Alternatively, I could just be careful about the
error paths.

  * We'd need to adapt this and make sure it works with whatever scheme
we decide is best for finding the right library. I suspect this would
just be adding another parameter to get_icu_library (and the hook) to
represent the new collation provider Oid (and get_icu_library could use
that to look up the library names and load them).

Comments welcome.

[1]
https://www.postgresql.org/message-id/[email protected]


-- 
Jeff Davis
PostgreSQL Contributor Team - AWS




Attachments:

  [text/x-patch] v2-0001-Add-pg_strcoll-and-pg_strncoll.patch (19.3K, ../../[email protected]/2-v2-0001-Add-pg_strcoll-and-pg_strncoll.patch)
  download | inline diff:
From 5d144e19686a0ae01e1d80a9f8c356079f41b9eb Mon Sep 17 00:00:00 2001
From: Jeff Davis <[email protected]>
Date: Thu, 1 Dec 2022 14:45:15 -0800
Subject: [PATCH v2 1/6] Add pg_strcoll() and pg_strncoll().

Callers with NUL-terminated strings should call the former; callers
with strings and their length should call the latter.
---
 src/backend/utils/adt/pg_locale.c | 420 ++++++++++++++++++++++++++++--
 src/backend/utils/adt/varlena.c   | 230 +---------------
 src/include/utils/pg_locale.h     |   3 +
 3 files changed, 406 insertions(+), 247 deletions(-)

diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 2b42d9ccd8..6cd629ecb4 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -79,6 +79,12 @@
 #include <shlwapi.h>
 #endif
 
+/*
+ * This should be large enough that most strings will fit, but small enough
+ * that we feel comfortable putting it on the stack
+ */
+#define		TEXTBUFLEN			1024
+
 #define		MAX_L10N_DATA		80
 
 
@@ -123,6 +129,19 @@ static char *IsoLocaleName(const char *);
 #endif
 
 #ifdef USE_ICU
+/*
+ * Converter object for converting between ICU's UChar strings and C strings
+ * in database encoding.  Since the database encoding doesn't change, we only
+ * need one of these per session.
+ */
+static UConverter *icu_converter = NULL;
+
+static void init_icu_converter(void);
+static size_t uchar_length(UConverter *converter,
+						   const char *str, size_t len);
+static int32_t uchar_convert(UConverter *converter,
+							 UChar *dest, int32_t destlen,
+							 const char *str, size_t srclen);
 static void icu_set_collation_attributes(UCollator *collator, const char *loc);
 #endif
 
@@ -1731,15 +1750,356 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 	return collversion;
 }
 
+/*
+ * pg_strncoll_libc_win32_utf8
+ *
+ * Win32 does not have UTF-8. Convert UTF8 arguments to wide characters and
+ * invoke wcscoll() or wcscoll_l().
+ */
+#ifdef WIN32
+static int
+pg_strncoll_libc_win32_utf8(const char *arg1, size_t len1, const char *arg2,
+							size_t len2, pg_locale_t locale)
+{
+	char		sbuf[TEXTBUFLEN];
+	char	   *buf = sbuf;
+	char	   *a1p,
+			   *a2p;
+	int			a1len = len1 * 2 + 2;
+	int			a2len = len2 * 2 + 2;
+	int			r;
+	int			result;
+
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+	Assert(GetDatabaseEncoding() == PG_UTF8);
+#ifndef WIN32
+	Assert(false);
+#endif
+
+	if (a1len + a2len > TEXTBUFLEN)
+		buf = palloc(a1len + a2len);
+
+	a1p = buf;
+	a2p = buf + a1len;
+
+	/* API does not work for zero-length input */
+	if (len1 == 0)
+		r = 0;
+	else
+	{
+		r = MultiByteToWideChar(CP_UTF8, 0, arg1, len1,
+								(LPWSTR) a1p, a1len / 2);
+		if (!r)
+			ereport(ERROR,
+					(errmsg("could not convert string to UTF-16: error code %lu",
+							GetLastError())));
+	}
+	((LPWSTR) a1p)[r] = 0;
+
+	if (len2 == 0)
+		r = 0;
+	else
+	{
+		r = MultiByteToWideChar(CP_UTF8, 0, arg2, len2,
+								(LPWSTR) a2p, a2len / 2);
+		if (!r)
+			ereport(ERROR,
+					(errmsg("could not convert string to UTF-16: error code %lu",
+							GetLastError())));
+	}
+	((LPWSTR) a2p)[r] = 0;
+
+	errno = 0;
+#ifdef HAVE_LOCALE_T
+	if (locale)
+		result = wcscoll_l((LPWSTR) a1p, (LPWSTR) a2p, locale->info.lt);
+	else
+#endif
+		result = wcscoll((LPWSTR) a1p, (LPWSTR) a2p);
+	if (result == 2147483647)	/* _NLSCMPERROR; missing from mingw
+								 * headers */
+		ereport(ERROR,
+				(errmsg("could not compare Unicode strings: %m")));
+
+	if (buf != sbuf)
+		pfree(buf);
+
+	return result;
+}
+#endif							/* WIN32 */
+
+/*
+ * pg_strcoll_libc
+ *
+ * Call strcoll(), strcoll_l(), wcscoll(), or wcscoll_l() as appropriate for
+ * the given locale, platform, and database encoding. If the locale is NULL,
+ * use the database collation.
+ *
+ * Arguments must be encoded in the database encoding and nul-terminated.
+ */
+static int
+pg_strcoll_libc(const char *arg1, const char *arg2, pg_locale_t locale)
+{
+	int result;
+
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+#ifdef WIN32
+	if (GetDatabaseEncoding() == PG_UTF8)
+	{
+		size_t len1 = strlen(arg1);
+		size_t len2 = strlen(arg2);
+		result = pg_strncoll_libc_win32_utf8(arg1, len1, arg2, len2, locale);
+	}
+	else
+#endif							/* WIN32 */
+	if (locale)
+	{
+#ifdef HAVE_LOCALE_T
+		result = strcoll_l(arg1, arg2, locale->info.lt);
+#else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+#endif
+	}
+	else
+		result = strcoll(arg1, arg2);
+
+	return result;
+}
+
+/*
+ * pg_strncoll_libc
+ *
+ * Null-terminate the arguments and call pg_strcoll_libc().
+ */
+static int
+pg_strncoll_libc(const char *arg1, size_t len1, const char *arg2, size_t len2,
+				 pg_locale_t locale)
+{
+	char	 sbuf[TEXTBUFLEN];
+	char	*buf	  = sbuf;
+	size_t	 bufsize1 = len1 + 1;
+	size_t	 bufsize2 = len2 + 1;
+	char	*arg1n;
+	char	*arg2n;
+	int		 result;
+
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+
+#ifdef WIN32
+	/* check for this case before doing the work for nul-termination */
+	if (GetDatabaseEncoding() == PG_UTF8)
+		return pg_strncoll_libc_win32_utf8(arg1, len1, arg2, len2, locale);
+#endif							/* WIN32 */
+
+	if (bufsize1 + bufsize2 > TEXTBUFLEN)
+		buf = palloc(bufsize1 + bufsize2);
+
+	arg1n = buf;
+	arg2n = buf + bufsize1;
+
+	/* nul-terminate arguments */
+	memcpy(arg1n, arg1, len1);
+	arg1n[len1] = '\0';
+	memcpy(arg2n, arg2, len2);
+	arg2n[len2] = '\0';
+
+	result = pg_strcoll_libc(arg1n, arg2n, locale);
+
+	if (buf != sbuf)
+		pfree(buf);
+
+	return result;
+}
 
 #ifdef USE_ICU
+
 /*
- * Converter object for converting between ICU's UChar strings and C strings
- * in database encoding.  Since the database encoding doesn't change, we only
- * need one of these per session.
+ * pg_strncoll_icu_no_utf8
+ *
+ * Convert the arguments from the database encoding to UChar strings, then
+ * call ucol_strcoll().
+ *
+ * When the database encoding is UTF-8, and ICU supports ucol_strcollUTF8(),
+ * caller should call that instead.
  */
-static UConverter *icu_converter = NULL;
+static int
+pg_strncoll_icu_no_utf8(const char *arg1, size_t len1,
+						const char *arg2, size_t len2, pg_locale_t locale)
+{
+	char	 sbuf[TEXTBUFLEN];
+	char	*buf = sbuf;
+	int32_t	 ulen1;
+	int32_t	 ulen2;
+	size_t   bufsize1;
+	size_t   bufsize2;
+	UChar	*uchar1,
+			*uchar2;
+	int		 result;
+
+	Assert(locale->provider == COLLPROVIDER_ICU);
+#ifdef HAVE_UCOL_STRCOLLUTF8
+	Assert(GetDatabaseEncoding() != PG_UTF8);
+#endif
+
+	init_icu_converter();
+
+	ulen1 = uchar_length(icu_converter, arg1, len1);
+	ulen2 = uchar_length(icu_converter, arg2, len2);
+
+	bufsize1 = (ulen1 + 1) * sizeof(UChar);
+	bufsize2 = (ulen2 + 1) * sizeof(UChar);
+
+	if (bufsize1 + bufsize2 > TEXTBUFLEN)
+		buf = palloc(bufsize1 + bufsize2);
+
+	uchar1 = (UChar *) buf;
+	uchar2 = (UChar *) (buf + bufsize1);
 
+	ulen1 = uchar_convert(icu_converter, uchar1, ulen1 + 1, arg1, len1);
+	ulen2 = uchar_convert(icu_converter, uchar2, ulen2 + 1, arg2, len2);
+
+	result = ucol_strcoll(locale->info.icu.ucol,
+						  uchar1, ulen1,
+						  uchar2, ulen2);
+
+	if (buf != sbuf)
+		pfree(buf);
+
+	return result;
+}
+
+/*
+ * pg_strncoll_icu
+ *
+ * Call ucol_strcollUTF8() or ucol_strcoll() as appropriate for the given
+ * database encoding.
+ *
+ * Arguments must be encoded in the database encoding.
+ */
+static int
+pg_strncoll_icu(const char *arg1, size_t len1, const char *arg2, size_t len2,
+				pg_locale_t locale)
+{
+	int result;
+
+	Assert(locale->provider == COLLPROVIDER_ICU);
+
+#ifdef HAVE_UCOL_STRCOLLUTF8
+	if (GetDatabaseEncoding() == PG_UTF8)
+	{
+		UErrorCode	status;
+
+		status = U_ZERO_ERROR;
+		result = ucol_strcollUTF8(locale->info.icu.ucol,
+								  arg1, len1,
+								  arg2, len2,
+								  &status);
+		if (U_FAILURE(status))
+			ereport(ERROR,
+					(errmsg("collation failed: %s", u_errorName(status))));
+	}
+	else
+#endif
+	{
+		result = pg_strncoll_icu_no_utf8(arg1, len1, arg2, len2, locale);
+	}
+
+	return result;
+}
+
+/*
+ * pg_strcoll_icu
+ *
+ * Calculate the string lengths and call pg_strncoll_icu().
+ */
+static int
+pg_strcoll_icu(const char *arg1, const char *arg2, pg_locale_t locale)
+{
+	Assert(locale->provider == COLLPROVIDER_ICU);
+	return pg_strncoll_icu(arg1, -1, arg2, -1, locale);
+}
+
+#endif							/* USE_ICU */
+
+/*
+ * pg_strcoll
+ *
+ * Call ucol_strcollUTF8(), ucol_strcoll(), strcoll(), strcoll_l(), wcscoll(),
+ * or wcscoll_l() as appropriate for the given locale, platform, and database
+ * encoding. If the locale is not specified, use the database collation.
+ *
+ * Arguments must be encoded in the database encoding and nul-terminated.
+ *
+ * If the collation is deterministic, break ties with strcmp().
+ */
+int
+pg_strcoll(const char *arg1, const char *arg2, pg_locale_t locale)
+{
+	int			result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strcoll_libc(arg1, arg2, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strcoll_icu(arg1, arg2, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	/* Break tie if necessary. */
+	if (result == 0 && (!locale || locale->deterministic))
+		result = strcmp(arg1, arg2);
+
+	return result;
+}
+
+/*
+ * pg_strncoll
+ *
+ * Call ucol_strcollUTF8(), ucol_strcoll(), strcoll(), strcoll_l(), wcscoll(),
+ * or wcscoll_l() as appropriate for the given locale, platform, and database
+ * encoding. If the locale is not specified, use the database collation.
+ *
+ * Arguments must be encoded in the database encoding.
+ *
+ * If the collation is deterministic, break ties with memcmp(), and then with
+ * the string length.
+ *
+ * This function may need to nul-terminate the arguments for libc functions;
+ * so if the caller already has nul-terminated strings, it should call
+ * pg_strcoll() instead.
+ */
+int
+pg_strncoll(const char *arg1, size_t len1, const char *arg2, size_t len2,
+			pg_locale_t locale)
+{
+	int		 result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strncoll_libc(arg1, len1, arg2, len2, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strncoll_icu(arg1, len1, arg2, len2, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	/* Break tie if necessary. */
+	if (result == 0 && (!locale || locale->deterministic))
+	{
+		result = memcmp(arg1, arg2, Min(len1, len2));
+		if ((result == 0) && (len1 != len2))
+			result = (len1 < len2) ? -1 : 1;
+	}
+
+	return result;
+}
+
+
+#ifdef USE_ICU
 static void
 init_icu_converter(void)
 {
@@ -1767,6 +2127,39 @@ init_icu_converter(void)
 	icu_converter = conv;
 }
 
+/*
+ * Find length, in UChars, of given string if converted to UChar string.
+ */
+static size_t
+uchar_length(UConverter *converter, const char *str, size_t len)
+{
+	UErrorCode	status = U_ZERO_ERROR;
+	int32_t		ulen;
+	ulen = ucnv_toUChars(converter, NULL, 0, str, len, &status);
+	if (U_FAILURE(status) && status != U_BUFFER_OVERFLOW_ERROR)
+		ereport(ERROR,
+				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+	return ulen;
+}
+
+/*
+ * Convert the given source string into a UChar string, stored in dest, and
+ * return the length (in UChars).
+ */
+static int32_t
+uchar_convert(UConverter *converter, UChar *dest, int32_t destlen,
+			  const char *src, size_t srclen)
+{
+	UErrorCode	status = U_ZERO_ERROR;
+	int32_t		ulen;
+	status = U_ZERO_ERROR;
+	ulen = ucnv_toUChars(converter, dest, destlen, src, srclen, &status);
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+	return ulen;
+}
+
 /*
  * Convert a string in the database encoding into a string of UChars.
  *
@@ -1782,26 +2175,15 @@ init_icu_converter(void)
 int32_t
 icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes)
 {
-	UErrorCode	status;
-	int32_t		len_uchar;
+	int32_t len_uchar;
 
 	init_icu_converter();
 
-	status = U_ZERO_ERROR;
-	len_uchar = ucnv_toUChars(icu_converter, NULL, 0,
-							  buff, nbytes, &status);
-	if (U_FAILURE(status) && status != U_BUFFER_OVERFLOW_ERROR)
-		ereport(ERROR,
-				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+	len_uchar = uchar_length(icu_converter, buff, nbytes);
 
 	*buff_uchar = palloc((len_uchar + 1) * sizeof(**buff_uchar));
-
-	status = U_ZERO_ERROR;
-	len_uchar = ucnv_toUChars(icu_converter, *buff_uchar, len_uchar + 1,
-							  buff, nbytes, &status);
-	if (U_FAILURE(status))
-		ereport(ERROR,
-				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+	len_uchar = uchar_convert(icu_converter,
+							  *buff_uchar, len_uchar + 1, buff, nbytes);
 
 	return len_uchar;
 }
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c5e7ee7ca2..c904bc0825 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1535,10 +1535,6 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 	}
 	else
 	{
-		char		a1buf[TEXTBUFLEN];
-		char		a2buf[TEXTBUFLEN];
-		char	   *a1p,
-				   *a2p;
 		pg_locale_t mylocale;
 
 		mylocale = pg_newlocale_from_collation(collid);
@@ -1555,171 +1551,7 @@ varstr_cmp(const char *arg1, int len1, const char *arg2, int len2, Oid collid)
 		if (len1 == len2 && memcmp(arg1, arg2, len1) == 0)
 			return 0;
 
-#ifdef WIN32
-		/* Win32 does not have UTF-8, so we need to map to UTF-16 */
-		if (GetDatabaseEncoding() == PG_UTF8
-			&& (!mylocale || mylocale->provider == COLLPROVIDER_LIBC))
-		{
-			int			a1len;
-			int			a2len;
-			int			r;
-
-			if (len1 >= TEXTBUFLEN / 2)
-			{
-				a1len = len1 * 2 + 2;
-				a1p = palloc(a1len);
-			}
-			else
-			{
-				a1len = TEXTBUFLEN;
-				a1p = a1buf;
-			}
-			if (len2 >= TEXTBUFLEN / 2)
-			{
-				a2len = len2 * 2 + 2;
-				a2p = palloc(a2len);
-			}
-			else
-			{
-				a2len = TEXTBUFLEN;
-				a2p = a2buf;
-			}
-
-			/* stupid Microsloth API does not work for zero-length input */
-			if (len1 == 0)
-				r = 0;
-			else
-			{
-				r = MultiByteToWideChar(CP_UTF8, 0, arg1, len1,
-										(LPWSTR) a1p, a1len / 2);
-				if (!r)
-					ereport(ERROR,
-							(errmsg("could not convert string to UTF-16: error code %lu",
-									GetLastError())));
-			}
-			((LPWSTR) a1p)[r] = 0;
-
-			if (len2 == 0)
-				r = 0;
-			else
-			{
-				r = MultiByteToWideChar(CP_UTF8, 0, arg2, len2,
-										(LPWSTR) a2p, a2len / 2);
-				if (!r)
-					ereport(ERROR,
-							(errmsg("could not convert string to UTF-16: error code %lu",
-									GetLastError())));
-			}
-			((LPWSTR) a2p)[r] = 0;
-
-			errno = 0;
-#ifdef HAVE_LOCALE_T
-			if (mylocale)
-				result = wcscoll_l((LPWSTR) a1p, (LPWSTR) a2p, mylocale->info.lt);
-			else
-#endif
-				result = wcscoll((LPWSTR) a1p, (LPWSTR) a2p);
-			if (result == 2147483647)	/* _NLSCMPERROR; missing from mingw
-										 * headers */
-				ereport(ERROR,
-						(errmsg("could not compare Unicode strings: %m")));
-
-			/* Break tie if necessary. */
-			if (result == 0 &&
-				(!mylocale || mylocale->deterministic))
-			{
-				result = memcmp(arg1, arg2, Min(len1, len2));
-				if ((result == 0) && (len1 != len2))
-					result = (len1 < len2) ? -1 : 1;
-			}
-
-			if (a1p != a1buf)
-				pfree(a1p);
-			if (a2p != a2buf)
-				pfree(a2p);
-
-			return result;
-		}
-#endif							/* WIN32 */
-
-		if (len1 >= TEXTBUFLEN)
-			a1p = (char *) palloc(len1 + 1);
-		else
-			a1p = a1buf;
-		if (len2 >= TEXTBUFLEN)
-			a2p = (char *) palloc(len2 + 1);
-		else
-			a2p = a2buf;
-
-		memcpy(a1p, arg1, len1);
-		a1p[len1] = '\0';
-		memcpy(a2p, arg2, len2);
-		a2p[len2] = '\0';
-
-		if (mylocale)
-		{
-			if (mylocale->provider == COLLPROVIDER_ICU)
-			{
-#ifdef USE_ICU
-#ifdef HAVE_UCOL_STRCOLLUTF8
-				if (GetDatabaseEncoding() == PG_UTF8)
-				{
-					UErrorCode	status;
-
-					status = U_ZERO_ERROR;
-					result = ucol_strcollUTF8(mylocale->info.icu.ucol,
-											  arg1, len1,
-											  arg2, len2,
-											  &status);
-					if (U_FAILURE(status))
-						ereport(ERROR,
-								(errmsg("collation failed: %s", u_errorName(status))));
-				}
-				else
-#endif
-				{
-					int32_t		ulen1,
-								ulen2;
-					UChar	   *uchar1,
-							   *uchar2;
-
-					ulen1 = icu_to_uchar(&uchar1, arg1, len1);
-					ulen2 = icu_to_uchar(&uchar2, arg2, len2);
-
-					result = ucol_strcoll(mylocale->info.icu.ucol,
-										  uchar1, ulen1,
-										  uchar2, ulen2);
-
-					pfree(uchar1);
-					pfree(uchar2);
-				}
-#else							/* not USE_ICU */
-				/* shouldn't happen */
-				elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
-#endif							/* not USE_ICU */
-			}
-			else
-			{
-#ifdef HAVE_LOCALE_T
-				result = strcoll_l(a1p, a2p, mylocale->info.lt);
-#else
-				/* shouldn't happen */
-				elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
-#endif
-			}
-		}
-		else
-			result = strcoll(a1p, a2p);
-
-		/* Break tie if necessary. */
-		if (result == 0 &&
-			(!mylocale || mylocale->deterministic))
-			result = strcmp(a1p, a2p);
-
-		if (a1p != a1buf)
-			pfree(a1p);
-		if (a2p != a2buf)
-			pfree(a2p);
+		result = pg_strncoll(arg1, len1, arg2, len2, mylocale);
 	}
 
 	return result;
@@ -2377,65 +2209,7 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 		return sss->last_returned;
 	}
 
-	if (sss->locale)
-	{
-		if (sss->locale->provider == COLLPROVIDER_ICU)
-		{
-#ifdef USE_ICU
-#ifdef HAVE_UCOL_STRCOLLUTF8
-			if (GetDatabaseEncoding() == PG_UTF8)
-			{
-				UErrorCode	status;
-
-				status = U_ZERO_ERROR;
-				result = ucol_strcollUTF8(sss->locale->info.icu.ucol,
-										  a1p, len1,
-										  a2p, len2,
-										  &status);
-				if (U_FAILURE(status))
-					ereport(ERROR,
-							(errmsg("collation failed: %s", u_errorName(status))));
-			}
-			else
-#endif
-			{
-				int32_t		ulen1,
-							ulen2;
-				UChar	   *uchar1,
-						   *uchar2;
-
-				ulen1 = icu_to_uchar(&uchar1, a1p, len1);
-				ulen2 = icu_to_uchar(&uchar2, a2p, len2);
-
-				result = ucol_strcoll(sss->locale->info.icu.ucol,
-									  uchar1, ulen1,
-									  uchar2, ulen2);
-
-				pfree(uchar1);
-				pfree(uchar2);
-			}
-#else							/* not USE_ICU */
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", sss->locale->provider);
-#endif							/* not USE_ICU */
-		}
-		else
-		{
-#ifdef HAVE_LOCALE_T
-			result = strcoll_l(sss->buf1, sss->buf2, sss->locale->info.lt);
-#else
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", sss->locale->provider);
-#endif
-		}
-	}
-	else
-		result = strcoll(sss->buf1, sss->buf2);
-
-	/* Break tie if necessary. */
-	if (result == 0 &&
-		(!sss->locale || sss->locale->deterministic))
-		result = strcmp(sss->buf1, sss->buf2);
+	result = pg_strcoll(sss->buf1, sss->buf2, sss->locale);
 
 	/* Cache result, perhaps saving an expensive strcoll() call next time */
 	sss->cache_blob = false;
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index a875942123..bf70ae08ca 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -100,6 +100,9 @@ extern void make_icu_collator(const char *iculocstr,
 extern pg_locale_t pg_newlocale_from_collation(Oid collid);
 
 extern char *get_collation_actual_version(char collprovider, const char *collcollate);
+extern int pg_strcoll(const char *arg1, const char *arg2, pg_locale_t locale);
+extern int pg_strncoll(const char *arg1, size_t len1,
+					   const char *arg2, size_t len2, pg_locale_t locale);
 
 #ifdef USE_ICU
 extern int32_t icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes);
-- 
2.34.1



  [text/x-patch] v2-0002-Add-pg_strxfrm-and-pg_strxfrm_prefix.patch (24.2K, ../../[email protected]/3-v2-0002-Add-pg_strxfrm-and-pg_strxfrm_prefix.patch)
  download | inline diff:
From bbee706cfa8a074c31429b178242c77084af9d6c Mon Sep 17 00:00:00 2001
From: Jeff Davis <[email protected]>
Date: Thu, 1 Dec 2022 14:41:38 -0800
Subject: [PATCH v2 2/6] Add pg_strxfrm() and pg_strxfrm_prefix().

Callers with a NUL-terminated string should call the former; callers
with a string and length should call the latter.
---
 src/backend/access/hash/hashfunc.c |  45 ++--
 src/backend/utils/adt/pg_locale.c  | 382 +++++++++++++++++++++++++++++
 src/backend/utils/adt/varchar.c    |  41 ++--
 src/backend/utils/adt/varlena.c    | 142 +++--------
 src/include/utils/pg_locale.h      |  10 +
 5 files changed, 470 insertions(+), 150 deletions(-)

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index f890f79ee1..b8136e496f 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -291,21 +291,19 @@ hashtext(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 		if (mylocale->provider == COLLPROVIDER_ICU)
 		{
-			int32_t		ulen = -1;
-			UChar	   *uchar = NULL;
-			Size		bsize;
-			uint8_t    *buf;
+			Size		bsize, rsize;
+			char	   *buf;
+			const char *keydata = VARDATA_ANY(key);
+			size_t		keylen = VARSIZE_ANY_EXHDR(key);
 
-			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
-
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
-			pfree(uchar);
 
-			result = hash_any(buf, bsize);
+			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+			if (rsize != bsize)
+				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+			result = hash_any((uint8_t *) buf, bsize);
 
 			pfree(buf);
 		}
@@ -349,21 +347,20 @@ hashtextextended(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 		if (mylocale->provider == COLLPROVIDER_ICU)
 		{
-			int32_t		ulen = -1;
-			UChar	   *uchar = NULL;
-			Size		bsize;
-			uint8_t    *buf;
+			Size		bsize, rsize;
+			char	   *buf;
+			const char *keydata = VARDATA_ANY(key);
+			size_t		keylen = VARSIZE_ANY_EXHDR(key);
 
-			ulen = icu_to_uchar(&uchar, VARDATA_ANY(key), VARSIZE_ANY_EXHDR(key));
-
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
-			pfree(uchar);
 
-			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
+			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+			if (rsize != bsize)
+				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+			result = hash_any_extended((uint8_t *) buf, bsize,
+									   PG_GETARG_INT64(1));
 
 			pfree(buf);
 		}
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 6cd629ecb4..133bb03a13 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -2099,6 +2099,388 @@ pg_strncoll(const char *arg1, size_t len1, const char *arg2, size_t len2,
 }
 
 
+static size_t
+pg_strxfrm_libc(char *dest, const char *src, size_t destsize,
+				pg_locale_t locale)
+{
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+
+#ifdef TRUST_STXFRM
+#ifdef HAVE_LOCALE_T
+	if (locale)
+		return strxfrm_l(dest, src, destsize, locale->info.lt);
+	else
+#endif
+		return strxfrm(dest, src, destsize);
+#else
+	/* shouldn't happen */
+	elog(ERROR, "unsupported collprovider: %c", locale->provider);
+#endif
+}
+
+static size_t
+pg_strnxfrm_libc(char *dest, const char *src, size_t srclen, size_t destsize,
+				 pg_locale_t locale)
+{
+	char	 sbuf[TEXTBUFLEN];
+	char	*buf	 = sbuf;
+	size_t	 bufsize = srclen + 1;
+	size_t	 result;
+
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+
+	if (bufsize > TEXTBUFLEN)
+		buf = palloc(bufsize);
+
+	/* nul-terminate arguments */
+	memcpy(buf, src, srclen);
+	buf[srclen] = '\0';
+
+	result = pg_strxfrm_libc(dest, buf, destsize, locale);
+
+	if (buf != sbuf)
+		pfree(buf);
+
+	return result;
+}
+
+static size_t
+pg_strxfrm_prefix_libc(char *dest, const char *src, size_t destsize,
+					   pg_locale_t locale)
+{
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+	/* unsupported; shouldn't happen */
+	elog(ERROR, "collprovider '%c' does not support pg_strxfrm_prefix()",
+		 locale->provider);
+}
+
+static size_t
+pg_strnxfrm_prefix_libc(char *dest, const char *src, size_t srclen,
+						size_t destsize, pg_locale_t locale)
+{
+	Assert(!locale || locale->provider == COLLPROVIDER_LIBC);
+	/* unsupported; shouldn't happen */
+	elog(ERROR, "collprovider '%c' does not support pg_strnxfrm_prefix()",
+		 locale->provider);
+}
+
+#ifdef USE_ICU
+
+static size_t
+pg_strnxfrm_icu(char *dest, const char *src, size_t srclen, size_t destsize,
+				pg_locale_t locale)
+{
+	char	 sbuf[TEXTBUFLEN];
+	char	*buf	= sbuf;
+	UChar	*uchar;
+	int32_t	 ulen;
+	size_t   uchar_bsize;
+	Size	 result_bsize;
+
+	Assert(locale->provider == COLLPROVIDER_ICU);
+
+	init_icu_converter();
+
+	ulen = uchar_length(icu_converter, src, srclen);
+
+	uchar_bsize = (ulen + 1) * sizeof(UChar);
+
+	if (uchar_bsize > TEXTBUFLEN)
+		buf = palloc(uchar_bsize);
+
+	uchar = (UChar *) buf;
+
+	ulen = uchar_convert(icu_converter, uchar, ulen + 1, src, srclen);
+
+	result_bsize = ucol_getSortKey(locale->info.icu.ucol,
+								   uchar, ulen,
+								   (uint8_t *) dest, destsize);
+
+	if (buf != sbuf)
+		pfree(buf);
+
+	return result_bsize;
+}
+
+static size_t
+pg_strxfrm_icu(char *dest, const char *src, size_t destsize,
+			   pg_locale_t locale)
+{
+	Assert(locale->provider == COLLPROVIDER_ICU);
+	return pg_strnxfrm_icu(dest, src, -1, destsize, locale);
+}
+
+static size_t
+pg_strnxfrm_prefix_icu_no_utf8(char *dest, const char *src, size_t srclen,
+							   size_t destsize, pg_locale_t locale)
+{
+	char			 sbuf[TEXTBUFLEN];
+	char			*buf   = sbuf;
+	UCharIterator	 iter;
+	uint32_t		 state[2];
+	UErrorCode		 status;
+	int32_t			 ulen  = -1;
+	UChar			*uchar = NULL;
+	size_t			 uchar_bsize;
+	Size			 result_bsize;
+
+	Assert(locale->provider == COLLPROVIDER_ICU);
+	Assert(GetDatabaseEncoding() != PG_UTF8);
+
+	init_icu_converter();
+
+	ulen = uchar_length(icu_converter, src, srclen);
+
+	uchar_bsize = (ulen + 1) * sizeof(UChar);
+
+	if (uchar_bsize > TEXTBUFLEN)
+		buf = palloc(uchar_bsize);
+
+	uchar = (UChar *) buf;
+
+	ulen = uchar_convert(icu_converter, uchar, ulen + 1, src, srclen);
+
+	uiter_setString(&iter, uchar, ulen);
+	state[0] = state[1] = 0;	/* won't need that again */
+	status = U_ZERO_ERROR;
+	result_bsize = ucol_nextSortKeyPart(locale->info.icu.ucol,
+										&iter,
+										state,
+										(uint8_t *) dest,
+										destsize,
+										&status);
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("sort key generation failed: %s",
+						u_errorName(status))));
+
+	return result_bsize;
+}
+
+static size_t
+pg_strnxfrm_prefix_icu(char *dest, const char *src, size_t srclen,
+					   size_t destsize, pg_locale_t locale)
+{
+	size_t result;
+
+	Assert(locale->provider == COLLPROVIDER_ICU);
+
+	if (GetDatabaseEncoding() == PG_UTF8)
+	{
+		UCharIterator iter;
+		uint32_t	state[2];
+		UErrorCode	status;
+
+		uiter_setUTF8(&iter, src, srclen);
+		state[0] = state[1] = 0;	/* won't need that again */
+		status = U_ZERO_ERROR;
+		result = ucol_nextSortKeyPart(locale->info.icu.ucol,
+									  &iter,
+									  state,
+									  (uint8_t *) dest,
+									  destsize,
+									  &status);
+		if (U_FAILURE(status))
+			ereport(ERROR,
+					(errmsg("sort key generation failed: %s",
+							u_errorName(status))));
+	}
+	else
+		result = pg_strnxfrm_prefix_icu_no_utf8(dest, src, srclen, destsize,
+												locale);
+
+	return result;
+}
+
+static size_t
+pg_strxfrm_prefix_icu(char *dest, const char *src, size_t destsize,
+					  pg_locale_t locale)
+{
+	Assert(locale->provider == COLLPROVIDER_ICU);
+	return pg_strnxfrm_prefix_icu(dest, src, -1, destsize, locale);
+}
+
+#endif
+
+/*
+ * Return true if the collation provider supports pg_strxfrm() and
+ * pg_strnxfrm(); otherwise false.
+ *
+ * Unfortunately, it seems that strxfrm() for non-C collations is broken on
+ * many common platforms; testing of multiple versions of glibc reveals that,
+ * for many locales, strcoll() and strxfrm() do not return consistent
+ * results. While no other libc other than Cygwin has so far been shown to
+ * have a problem, we take the conservative course of action for right now and
+ * disable this categorically.  (Users who are certain this isn't a problem on
+ * their system can define TRUST_STRXFRM.)
+ */
+bool
+pg_strxfrm_enabled(pg_locale_t locale)
+{
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+	{
+#ifdef TRUST_STRXFRM
+		return true;
+#else
+		return false;
+#endif
+	}
+	else if (locale->provider == COLLPROVIDER_ICU)
+		return true;
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+}
+
+/*
+ * pg_strxfrm
+ *
+ * Transforms 'src' to a nul-terminated string stored in 'dest' such that
+ * ordinary strcmp() on transformed strings is equivalent to pg_strcoll() on
+ * untransformed strings.
+ *
+ * The provided 'src' must be nul-terminated.
+ *
+ * If destsize is large enough to hold the result, returns the number of bytes
+ * copied to 'dest'; otherwise, returns the number of bytes needed to hold the
+ * result and leaves the contents of 'dest' undefined. If destsize is zero,
+ * 'dest' may be NULL.
+ */
+size_t
+pg_strxfrm(char *dest, const char *src, size_t destsize, pg_locale_t locale)
+{
+	size_t result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strxfrm_libc(dest, src, destsize, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strxfrm_icu(dest, src, destsize, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	return result;
+}
+
+/*
+ * pg_strnxfrm
+ *
+ * Transforms 'src' to a nul-terminated string stored in 'dest' such that
+ * ordinary strcmp() on transformed strings is equivalent to pg_strcoll() on
+ * untransformed strings.
+ *
+ * If destsize is large enough to hold the result, returns the number of bytes
+ * copied to 'dest'; otherwise, returns the number of bytes needed to hold the
+ * result and leaves the contents of 'dest' undefined. If destsize is zero,
+ * 'dest' may be NULL.
+ *
+ * This function may need to nul-terminate the argument for libc functions;
+ * so if the caller already has a nul-terminated string, it should call
+ * pg_strxfrm() instead.
+ */
+size_t
+pg_strnxfrm(char *dest, size_t destsize, const char *src, size_t srclen,
+			pg_locale_t locale)
+{
+	size_t result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strnxfrm_libc(dest, src, srclen, destsize, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strnxfrm_icu(dest, src, srclen, destsize, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	return result;
+}
+
+/*
+ * Return true if the collation provider supports pg_strxfrm_prefix() and
+ * pg_strnxfrm_prefix(); otherwise false.
+ */
+bool
+pg_strxfrm_prefix_enabled(pg_locale_t locale)
+{
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		return false;
+	else if (locale->provider == COLLPROVIDER_ICU)
+		return true;
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+}
+
+/*
+ * pg_strxfrm_prefix
+ *
+ * Transforms 'src' to a byte sequence stored in 'dest' such that ordinary
+ * memcmp() on the byte sequence is equivalent to pg_strcoll() on
+ * untransformed strings. The result is not nul-terminated.
+ *
+ * The provided 'src' must be nul-terminated.
+ *
+ * If destsize is not large enough to hold the entire result, stores just the
+ * prefix in 'dest'. Returns the number of bytes actually copied to 'dest'.
+ */
+size_t
+pg_strxfrm_prefix(char *dest, const char *src, size_t destsize,
+				  pg_locale_t locale)
+{
+	size_t result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strxfrm_prefix_libc(dest, src, destsize, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strxfrm_prefix_icu(dest, src, destsize, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	return result;
+}
+
+/*
+ * pg_strnxfrm_prefix
+ *
+ * Transforms 'src' to a byte sequence stored in 'dest' such that ordinary
+ * memcmp() on the byte sequence is equivalent to pg_strcoll() on
+ * untransformed strings. The result is not nul-terminated.
+ *
+ * The provided 'src' must be nul-terminated.
+ *
+ * If destsize is not large enough to hold the entire result, stores just the
+ * prefix in 'dest'. Returns the number of bytes actually copied to 'dest'.
+ *
+ * This function may need to nul-terminate the argument for libc functions;
+ * so if the caller already has a nul-terminated string, it should call
+ * pg_strxfrm_prefix() instead.
+ */
+size_t
+pg_strnxfrm_prefix(char *dest, size_t destsize, const char *src,
+				   size_t srclen, pg_locale_t locale)
+{
+	size_t result;
+
+	if (!locale || locale->provider == COLLPROVIDER_LIBC)
+		result = pg_strnxfrm_prefix_libc(dest, src, srclen, destsize, locale);
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+		result = pg_strnxfrm_prefix_icu(dest, src, srclen, destsize, locale);
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	return result;
+}
+
 #ifdef USE_ICU
 static void
 init_icu_converter(void)
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index a63c498181..d0bc528e9f 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -1019,21 +1019,17 @@ hashbpchar(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 		if (mylocale->provider == COLLPROVIDER_ICU)
 		{
-			int32_t		ulen = -1;
-			UChar	   *uchar = NULL;
-			Size		bsize;
-			uint8_t    *buf;
+			Size		bsize, rsize;
+			char	   *buf;
 
-			ulen = icu_to_uchar(&uchar, keydata, keylen);
-
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
-			pfree(uchar);
 
-			result = hash_any(buf, bsize);
+			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+			if (rsize != bsize)
+				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+			result = hash_any((uint8_t *) buf, bsize);
 
 			pfree(buf);
 		}
@@ -1081,21 +1077,18 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 		if (mylocale->provider == COLLPROVIDER_ICU)
 		{
-			int32_t		ulen = -1;
-			UChar	   *uchar = NULL;
-			Size		bsize;
-			uint8_t    *buf;
+			Size		bsize, rsize;
+			char	   *buf;
 
-			ulen = icu_to_uchar(&uchar, keydata, keylen);
-
-			bsize = ucol_getSortKey(mylocale->info.icu.ucol,
-									uchar, ulen, NULL, 0);
+			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
 			buf = palloc(bsize);
-			ucol_getSortKey(mylocale->info.icu.ucol,
-							uchar, ulen, buf, bsize);
-			pfree(uchar);
 
-			result = hash_any_extended(buf, bsize, PG_GETARG_INT64(1));
+			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+			if (rsize != bsize)
+				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+			result = hash_any_extended((uint8_t *) buf, bsize,
+									   PG_GETARG_INT64(1));
 
 			pfree(buf);
 		}
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index c904bc0825..2dfba4b488 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1887,20 +1887,6 @@ varstr_sortsupport(SortSupport ssup, Oid typid, Oid collid)
 		 */
 		locale = pg_newlocale_from_collation(collid);
 
-		/*
-		 * There is a further exception on Windows.  When the database
-		 * encoding is UTF-8 and we are not using the C collation, complex
-		 * hacks are required.  We don't currently have a comparator that
-		 * handles that case, so we fall back on the slow method of having the
-		 * sort code invoke bttextcmp() (in the case of text) via the fmgr
-		 * trampoline.  ICU locales work just the same on Windows, however.
-		 */
-#ifdef WIN32
-		if (GetDatabaseEncoding() == PG_UTF8 &&
-			!(locale && locale->provider == COLLPROVIDER_ICU))
-			return;
-#endif
-
 		/*
 		 * We use varlenafastcmp_locale except for type NAME.
 		 */
@@ -1916,13 +1902,7 @@ varstr_sortsupport(SortSupport ssup, Oid typid, Oid collid)
 
 	/*
 	 * Unfortunately, it seems that abbreviation for non-C collations is
-	 * broken on many common platforms; testing of multiple versions of glibc
-	 * reveals that, for many locales, strcoll() and strxfrm() do not return
-	 * consistent results, which is fatal to this optimization.  While no
-	 * other libc other than Cygwin has so far been shown to have a problem,
-	 * we take the conservative course of action for right now and disable
-	 * this categorically.  (Users who are certain this isn't a problem on
-	 * their system can define TRUST_STRXFRM.)
+	 * broken on many common platforms; see pg_strxfrm_enabled().
 	 *
 	 * Even apart from the risk of broken locales, it's possible that there
 	 * are platforms where the use of abbreviated keys should be disabled at
@@ -1935,10 +1915,8 @@ varstr_sortsupport(SortSupport ssup, Oid typid, Oid collid)
 	 * categorically, we may still want or need to disable it for particular
 	 * platforms.
 	 */
-#ifndef TRUST_STRXFRM
-	if (!collate_c && !(locale && locale->provider == COLLPROVIDER_ICU))
+	if (!collate_c && !pg_strxfrm_enabled(locale))
 		abbreviate = false;
-#endif
 
 	/*
 	 * If we're using abbreviated keys, or if we're using a locale-aware
@@ -2227,6 +2205,7 @@ varstrfastcmp_locale(char *a1p, int len1, char *a2p, int len2, SortSupport ssup)
 static Datum
 varstr_abbrev_convert(Datum original, SortSupport ssup)
 {
+	const size_t max_prefix_bytes = sizeof(Datum);
 	VarStringSortSupport *sss = (VarStringSortSupport *) ssup->ssup_extra;
 	VarString  *authoritative = DatumGetVarStringPP(original);
 	char	   *authoritative_data = VARDATA_ANY(authoritative);
@@ -2239,7 +2218,7 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 
 	pres = (char *) &res;
 	/* memset(), so any non-overwritten bytes are NUL */
-	memset(pres, 0, sizeof(Datum));
+	memset(pres, 0, max_prefix_bytes);
 	len = VARSIZE_ANY_EXHDR(authoritative);
 
 	/* Get number of bytes, ignoring trailing spaces */
@@ -2274,14 +2253,10 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 	 * thing: explicitly consider string length.
 	 */
 	if (sss->collate_c)
-		memcpy(pres, authoritative_data, Min(len, sizeof(Datum)));
+		memcpy(pres, authoritative_data, Min(len, max_prefix_bytes));
 	else
 	{
 		Size		bsize;
-#ifdef USE_ICU
-		int32_t		ulen = -1;
-		UChar	   *uchar = NULL;
-#endif
 
 		/*
 		 * We're not using the C collation, so fall back on strxfrm or ICU
@@ -2299,7 +2274,7 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 		if (sss->last_len1 == len && sss->cache_blob &&
 			memcmp(sss->buf1, authoritative_data, len) == 0)
 		{
-			memcpy(pres, sss->buf2, Min(sizeof(Datum), sss->last_len2));
+			memcpy(pres, sss->buf2, Min(max_prefix_bytes, sss->last_len2));
 			/* No change affecting cardinality, so no hashing required */
 			goto done;
 		}
@@ -2307,81 +2282,49 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 		memcpy(sss->buf1, authoritative_data, len);
 
 		/*
-		 * Just like strcoll(), strxfrm() expects a NUL-terminated string. Not
-		 * necessary for ICU, but doesn't hurt.
+		 * pg_strxfrm() and pg_strxfrm_prefix expect NUL-terminated
+		 * strings.
 		 */
 		sss->buf1[len] = '\0';
 		sss->last_len1 = len;
 
-#ifdef USE_ICU
-		/* When using ICU and not UTF8, convert string to UChar. */
-		if (sss->locale && sss->locale->provider == COLLPROVIDER_ICU &&
-			GetDatabaseEncoding() != PG_UTF8)
-			ulen = icu_to_uchar(&uchar, sss->buf1, len);
-#endif
-
-		/*
-		 * Loop: Call strxfrm() or ucol_getSortKey(), possibly enlarge buffer,
-		 * and try again.  Both of these functions have the result buffer
-		 * content undefined if the result did not fit, so we need to retry
-		 * until everything fits, even though we only need the first few bytes
-		 * in the end.  When using ucol_nextSortKeyPart(), however, we only
-		 * ask for as many bytes as we actually need.
-		 */
-		for (;;)
+		if (pg_strxfrm_prefix_enabled(sss->locale))
 		{
-#ifdef USE_ICU
-			if (sss->locale && sss->locale->provider == COLLPROVIDER_ICU)
+			if (sss->buflen2 < max_prefix_bytes)
 			{
-				/*
-				 * When using UTF8, use the iteration interface so we only
-				 * need to produce as many bytes as we actually need.
-				 */
-				if (GetDatabaseEncoding() == PG_UTF8)
-				{
-					UCharIterator iter;
-					uint32_t	state[2];
-					UErrorCode	status;
-
-					uiter_setUTF8(&iter, sss->buf1, len);
-					state[0] = state[1] = 0;	/* won't need that again */
-					status = U_ZERO_ERROR;
-					bsize = ucol_nextSortKeyPart(sss->locale->info.icu.ucol,
-												 &iter,
-												 state,
-												 (uint8_t *) sss->buf2,
-												 Min(sizeof(Datum), sss->buflen2),
-												 &status);
-					if (U_FAILURE(status))
-						ereport(ERROR,
-								(errmsg("sort key generation failed: %s",
-										u_errorName(status))));
-				}
-				else
-					bsize = ucol_getSortKey(sss->locale->info.icu.ucol,
-											uchar, ulen,
-											(uint8_t *) sss->buf2, sss->buflen2);
+				sss->buflen2 = Max(max_prefix_bytes,
+								   Min(sss->buflen2 * 2, MaxAllocSize));
+				sss->buf2 = repalloc(sss->buf2, sss->buflen2);
 			}
-			else
-#endif
-#ifdef HAVE_LOCALE_T
-			if (sss->locale && sss->locale->provider == COLLPROVIDER_LIBC)
-				bsize = strxfrm_l(sss->buf2, sss->buf1,
-								  sss->buflen2, sss->locale->info.lt);
-			else
-#endif
-				bsize = strxfrm(sss->buf2, sss->buf1, sss->buflen2);
-
-			sss->last_len2 = bsize;
-			if (bsize < sss->buflen2)
-				break;
 
+			bsize = pg_strxfrm_prefix(sss->buf2, sss->buf1,
+									  max_prefix_bytes, sss->locale);
+		}
+		else
+		{
 			/*
-			 * Grow buffer and retry.
+			 * Loop: Call pg_strxfrm(), possibly enlarge buffer, and try
+			 * again.  The pg_strxfrm() function leaves the result buffer
+			 * content undefined if the result did not fit, so we need to
+			 * retry until everything fits, even though we only need the first
+			 * few bytes in the end.
 			 */
-			sss->buflen2 = Max(bsize + 1,
-							   Min(sss->buflen2 * 2, MaxAllocSize));
-			sss->buf2 = repalloc(sss->buf2, sss->buflen2);
+			for (;;)
+			{
+				bsize = pg_strxfrm(sss->buf2, sss->buf1, sss->buflen2,
+								   sss->locale);
+
+				sss->last_len2 = bsize;
+				if (bsize < sss->buflen2)
+					break;
+
+				/*
+				 * Grow buffer and retry.
+				 */
+				sss->buflen2 = Max(bsize + 1,
+								   Min(sss->buflen2 * 2, MaxAllocSize));
+				sss->buf2 = repalloc(sss->buf2, sss->buflen2);
+			}
 		}
 
 		/*
@@ -2393,12 +2336,7 @@ varstr_abbrev_convert(Datum original, SortSupport ssup)
 		 * (Actually, even if there were NUL bytes in the blob it would be
 		 * okay.  See remarks on bytea case above.)
 		 */
-		memcpy(pres, sss->buf2, Min(sizeof(Datum), bsize));
-
-#ifdef USE_ICU
-		if (uchar)
-			pfree(uchar);
-#endif
+		memcpy(pres, sss->buf2, Min(max_prefix_bytes, bsize));
 	}
 
 	/*
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index bf70ae08ca..ceab0d4307 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -103,6 +103,16 @@ extern char *get_collation_actual_version(char collprovider, const char *collcol
 extern int pg_strcoll(const char *arg1, const char *arg2, pg_locale_t locale);
 extern int pg_strncoll(const char *arg1, size_t len1,
 					   const char *arg2, size_t len2, pg_locale_t locale);
+extern bool pg_strxfrm_enabled(pg_locale_t locale);
+extern size_t pg_strxfrm(char *dest, const char *src, size_t destsize,
+						 pg_locale_t locale);
+extern size_t pg_strnxfrm(char *dest, size_t destsize, const char *src,
+						  size_t srclen, pg_locale_t locale);
+extern bool pg_strxfrm_prefix_enabled(pg_locale_t locale);
+extern size_t pg_strxfrm_prefix(char *dest, const char *src, size_t destsize,
+								pg_locale_t locale);
+extern size_t pg_strnxfrm_prefix(char *dest, size_t destsize, const char *src,
+								 size_t srclen, pg_locale_t locale);
 
 #ifdef USE_ICU
 extern int32_t icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes);
-- 
2.34.1



  [text/x-patch] v2-0003-Refactor-pg_locale_t-routines.patch (45.7K, ../../[email protected]/4-v2-0003-Refactor-pg_locale_t-routines.patch)
  download | inline diff:
From 0145d9050ba3feef163c1d4a13660f904167458a Mon Sep 17 00:00:00 2001
From: Jeff Davis <[email protected]>
Date: Mon, 5 Dec 2022 10:43:52 -0800
Subject: [PATCH v2 3/6] Refactor pg_locale_t routines.

  * add pg_locale_internal.h to hide pg_locale_struct
  * move info.lt into info.libc.lt to match icu
  * introduce init_default_locale()
  * introduce collation_version_default_locale()
  * introduce pg_locale_deterministic() accessor
  * make default_locale a static global in pg_locale.c
  * refactor pg_newlocale_from_collation() to use multiple static
    functions and avoid allocating in TopMemoryContext until
    necessary
  * refactor get_collation_actual_version() to use
    pg_newlocale()
---
 src/backend/access/hash/hashfunc.c     |  82 ++---
 src/backend/commands/collationcmds.c   |   1 +
 src/backend/regex/regc_pg_locale.c     |  45 +--
 src/backend/utils/adt/formatting.c     |  25 +-
 src/backend/utils/adt/like.c           |   3 +-
 src/backend/utils/adt/like_support.c   |   3 +-
 src/backend/utils/adt/pg_locale.c      | 459 +++++++++++++++++--------
 src/backend/utils/adt/varchar.c        |  62 ++--
 src/backend/utils/adt/varlena.c        |   8 +-
 src/backend/utils/init/postinit.c      |  30 +-
 src/include/utils/pg_locale.h          |  55 +--
 src/include/utils/pg_locale_internal.h |  68 ++++
 12 files changed, 508 insertions(+), 333 deletions(-)
 create mode 100644 src/include/utils/pg_locale_internal.h

diff --git a/src/backend/access/hash/hashfunc.c b/src/backend/access/hash/hashfunc.c
index b8136e496f..6d9f014c5b 100644
--- a/src/backend/access/hash/hashfunc.c
+++ b/src/backend/access/hash/hashfunc.c
@@ -281,36 +281,28 @@ hashtext(PG_FUNCTION_ARGS)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (!mylocale || mylocale->deterministic)
+	if (pg_locale_deterministic(mylocale))
 	{
 		result = hash_any((unsigned char *) VARDATA_ANY(key),
 						  VARSIZE_ANY_EXHDR(key));
 	}
 	else
 	{
-#ifdef USE_ICU
-		if (mylocale->provider == COLLPROVIDER_ICU)
-		{
-			Size		bsize, rsize;
-			char	   *buf;
-			const char *keydata = VARDATA_ANY(key);
-			size_t		keylen = VARSIZE_ANY_EXHDR(key);
-
-			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
-			buf = palloc(bsize);
-
-			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
-			if (rsize != bsize)
-				elog(ERROR, "pg_strnxfrm() returned unexpected result");
-
-			result = hash_any((uint8_t *) buf, bsize);
-
-			pfree(buf);
-		}
-		else
-#endif
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
+		Size		bsize, rsize;
+		char	   *buf;
+		const char *keydata = VARDATA_ANY(key);
+		size_t		keylen = VARSIZE_ANY_EXHDR(key);
+
+		bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
+		buf = palloc(bsize);
+
+		rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+		if (rsize != bsize)
+			elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+		result = hash_any((uint8_t *) buf, bsize);
+
+		pfree(buf);
 	}
 
 	/* Avoid leaking memory for toasted inputs */
@@ -336,7 +328,7 @@ hashtextextended(PG_FUNCTION_ARGS)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (!mylocale || mylocale->deterministic)
+	if (pg_locale_deterministic(mylocale))
 	{
 		result = hash_any_extended((unsigned char *) VARDATA_ANY(key),
 								   VARSIZE_ANY_EXHDR(key),
@@ -344,30 +336,22 @@ hashtextextended(PG_FUNCTION_ARGS)
 	}
 	else
 	{
-#ifdef USE_ICU
-		if (mylocale->provider == COLLPROVIDER_ICU)
-		{
-			Size		bsize, rsize;
-			char	   *buf;
-			const char *keydata = VARDATA_ANY(key);
-			size_t		keylen = VARSIZE_ANY_EXHDR(key);
-
-			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
-			buf = palloc(bsize);
-
-			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
-			if (rsize != bsize)
-				elog(ERROR, "pg_strnxfrm() returned unexpected result");
-
-			result = hash_any_extended((uint8_t *) buf, bsize,
-									   PG_GETARG_INT64(1));
-
-			pfree(buf);
-		}
-		else
-#endif
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
+		Size		bsize, rsize;
+		char	   *buf;
+		const char *keydata = VARDATA_ANY(key);
+		size_t		keylen = VARSIZE_ANY_EXHDR(key);
+
+		bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
+		buf = palloc(bsize);
+
+		rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+		if (rsize != bsize)
+			elog(ERROR, "pg_strnxfrm() returned unexpected result");
+
+		result = hash_any_extended((uint8_t *) buf, bsize,
+								   PG_GETARG_INT64(1));
+
+		pfree(buf);
 	}
 
 	PG_FREE_IF_COPY(key, 0);
diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index 81e54e0ce6..9e84da4891 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -36,6 +36,7 @@
 #include "utils/builtins.h"
 #include "utils/lsyscache.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 #include "utils/rel.h"
 #include "utils/syscache.h"
 
diff --git a/src/backend/regex/regc_pg_locale.c b/src/backend/regex/regc_pg_locale.c
index 02d462a659..ac05efb558 100644
--- a/src/backend/regex/regc_pg_locale.c
+++ b/src/backend/regex/regc_pg_locale.c
@@ -17,6 +17,7 @@
 
 #include "catalog/pg_collation.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 
 /*
  * To provide as much functionality as possible on a variety of platforms,
@@ -306,13 +307,13 @@ pg_wc_isdigit(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswdigit_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswdigit_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isdigit_l((unsigned char) c, pg_regex_locale->info.lt));
+					isdigit_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -342,13 +343,13 @@ pg_wc_isalpha(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswalpha_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswalpha_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isalpha_l((unsigned char) c, pg_regex_locale->info.lt));
+					isalpha_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -378,13 +379,13 @@ pg_wc_isalnum(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswalnum_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswalnum_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isalnum_l((unsigned char) c, pg_regex_locale->info.lt));
+					isalnum_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -423,13 +424,13 @@ pg_wc_isupper(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswupper_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswupper_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isupper_l((unsigned char) c, pg_regex_locale->info.lt));
+					isupper_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -459,13 +460,13 @@ pg_wc_islower(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswlower_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswlower_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					islower_l((unsigned char) c, pg_regex_locale->info.lt));
+					islower_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -495,13 +496,13 @@ pg_wc_isgraph(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswgraph_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswgraph_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isgraph_l((unsigned char) c, pg_regex_locale->info.lt));
+					isgraph_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -531,13 +532,13 @@ pg_wc_isprint(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswprint_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswprint_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isprint_l((unsigned char) c, pg_regex_locale->info.lt));
+					isprint_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -567,13 +568,13 @@ pg_wc_ispunct(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswpunct_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswpunct_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					ispunct_l((unsigned char) c, pg_regex_locale->info.lt));
+					ispunct_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -603,13 +604,13 @@ pg_wc_isspace(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return iswspace_l((wint_t) c, pg_regex_locale->info.lt);
+				return iswspace_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			return (c <= (pg_wchar) UCHAR_MAX &&
-					isspace_l((unsigned char) c, pg_regex_locale->info.lt));
+					isspace_l((unsigned char) c, pg_regex_locale->info.libc.lt));
 #endif
 			break;
 		case PG_REGEX_LOCALE_ICU:
@@ -647,13 +648,13 @@ pg_wc_toupper(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return towupper_l((wint_t) c, pg_regex_locale->info.lt);
+				return towupper_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			if (c <= (pg_wchar) UCHAR_MAX)
-				return toupper_l((unsigned char) c, pg_regex_locale->info.lt);
+				return toupper_l((unsigned char) c, pg_regex_locale->info.libc.lt);
 #endif
 			return c;
 		case PG_REGEX_LOCALE_ICU:
@@ -691,13 +692,13 @@ pg_wc_tolower(pg_wchar c)
 		case PG_REGEX_LOCALE_WIDE_L:
 #ifdef HAVE_LOCALE_T
 			if (sizeof(wchar_t) >= 4 || c <= (pg_wchar) 0xFFFF)
-				return towlower_l((wint_t) c, pg_regex_locale->info.lt);
+				return towlower_l((wint_t) c, pg_regex_locale->info.libc.lt);
 #endif
 			/* FALL THRU */
 		case PG_REGEX_LOCALE_1BYTE_L:
 #ifdef HAVE_LOCALE_T
 			if (c <= (pg_wchar) UCHAR_MAX)
-				return tolower_l((unsigned char) c, pg_regex_locale->info.lt);
+				return tolower_l((unsigned char) c, pg_regex_locale->info.libc.lt);
 #endif
 			return c;
 		case PG_REGEX_LOCALE_ICU:
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index 26f498b5df..a4bc7fa5f5 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -87,6 +87,7 @@
 #include "utils/memutils.h"
 #include "utils/numeric.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 
 /* ----------
  * Convenience macros for error handling
@@ -1611,7 +1612,7 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
 	status = U_ZERO_ERROR;
 	len_dest = func(*buff_dest, len_dest, buff_source, len_source,
-					mylocale->info.icu.locale, &status);
+					mylocale->ctype, &status);
 	if (status == U_BUFFER_OVERFLOW_ERROR)
 	{
 		/* try again with adjusted length */
@@ -1619,7 +1620,7 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
 		status = U_ZERO_ERROR;
 		len_dest = func(*buff_dest, len_dest, buff_source, len_source,
-						mylocale->info.icu.locale, &status);
+						mylocale->ctype, &status);
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
@@ -1732,7 +1733,7 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 				{
 #ifdef HAVE_LOCALE_T
 					if (mylocale)
-						workspace[curr_char] = towlower_l(workspace[curr_char], mylocale->info.lt);
+						workspace[curr_char] = towlower_l(workspace[curr_char], mylocale->info.libc.lt);
 					else
 #endif
 						workspace[curr_char] = towlower(workspace[curr_char]);
@@ -1765,7 +1766,7 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 				{
 #ifdef HAVE_LOCALE_T
 					if (mylocale)
-						*p = tolower_l((unsigned char) *p, mylocale->info.lt);
+						*p = tolower_l((unsigned char) *p, mylocale->info.libc.lt);
 					else
 #endif
 						*p = pg_tolower((unsigned char) *p);
@@ -1854,7 +1855,7 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 				{
 #ifdef HAVE_LOCALE_T
 					if (mylocale)
-						workspace[curr_char] = towupper_l(workspace[curr_char], mylocale->info.lt);
+						workspace[curr_char] = towupper_l(workspace[curr_char], mylocale->info.libc.lt);
 					else
 #endif
 						workspace[curr_char] = towupper(workspace[curr_char]);
@@ -1887,7 +1888,7 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 				{
 #ifdef HAVE_LOCALE_T
 					if (mylocale)
-						*p = toupper_l((unsigned char) *p, mylocale->info.lt);
+						*p = toupper_l((unsigned char) *p, mylocale->info.libc.lt);
 					else
 #endif
 						*p = pg_toupper((unsigned char) *p);
@@ -1979,10 +1980,10 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 					if (mylocale)
 					{
 						if (wasalnum)
-							workspace[curr_char] = towlower_l(workspace[curr_char], mylocale->info.lt);
+							workspace[curr_char] = towlower_l(workspace[curr_char], mylocale->info.libc.lt);
 						else
-							workspace[curr_char] = towupper_l(workspace[curr_char], mylocale->info.lt);
-						wasalnum = iswalnum_l(workspace[curr_char], mylocale->info.lt);
+							workspace[curr_char] = towupper_l(workspace[curr_char], mylocale->info.libc.lt);
+						wasalnum = iswalnum_l(workspace[curr_char], mylocale->info.libc.lt);
 					}
 					else
 #endif
@@ -2024,10 +2025,10 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 					if (mylocale)
 					{
 						if (wasalnum)
-							*p = tolower_l((unsigned char) *p, mylocale->info.lt);
+							*p = tolower_l((unsigned char) *p, mylocale->info.libc.lt);
 						else
-							*p = toupper_l((unsigned char) *p, mylocale->info.lt);
-						wasalnum = isalnum_l((unsigned char) *p, mylocale->info.lt);
+							*p = toupper_l((unsigned char) *p, mylocale->info.libc.lt);
+						wasalnum = isalnum_l((unsigned char) *p, mylocale->info.libc.lt);
 					}
 					else
 #endif
diff --git a/src/backend/utils/adt/like.c b/src/backend/utils/adt/like.c
index 8e671b9fab..98714a0492 100644
--- a/src/backend/utils/adt/like.c
+++ b/src/backend/utils/adt/like.c
@@ -24,6 +24,7 @@
 #include "miscadmin.h"
 #include "utils/builtins.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 
 
 #define LIKE_TRUE						1
@@ -96,7 +97,7 @@ SB_lower_char(unsigned char c, pg_locale_t locale, bool locale_is_c)
 		return pg_ascii_tolower(c);
 #ifdef HAVE_LOCALE_T
 	else if (locale)
-		return tolower_l(c, locale->info.lt);
+		return tolower_l(c, locale->info.libc.lt);
 #endif
 	else
 		return pg_tolower(c);
diff --git a/src/backend/utils/adt/like_support.c b/src/backend/utils/adt/like_support.c
index 2d3aaaaf6b..28d23ac3ab 100644
--- a/src/backend/utils/adt/like_support.c
+++ b/src/backend/utils/adt/like_support.c
@@ -52,6 +52,7 @@
 #include "utils/datum.h"
 #include "utils/lsyscache.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 #include "utils/selfuncs.h"
 #include "utils/varlena.h"
 
@@ -1511,7 +1512,7 @@ pattern_char_isalpha(char c, bool is_multibyte,
 			(c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z');
 #ifdef HAVE_LOCALE_T
 	else if (locale && locale->provider == COLLPROVIDER_LIBC)
-		return isalpha_l((unsigned char) c, locale->info.lt);
+		return isalpha_l((unsigned char) c, locale->info.libc.lt);
 #endif
 	else
 		return isalpha((unsigned char) c);
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 133bb03a13..0a19845df4 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -65,6 +65,7 @@
 #include "utils/lsyscache.h"
 #include "utils/memutils.h"
 #include "utils/pg_locale.h"
+#include "utils/pg_locale_internal.h"
 #include "utils/syscache.h"
 
 #ifdef USE_ICU
@@ -128,6 +129,11 @@ static HTAB *collation_cache = NULL;
 static char *IsoLocaleName(const char *);
 #endif
 
+/*
+ * Database default locale.
+ */
+static pg_locale_t default_locale = NULL;
+
 #ifdef USE_ICU
 /*
  * Converter object for converting between ICU's UChar strings and C strings
@@ -1333,7 +1339,7 @@ lc_collate_is_c(Oid collation)
 		static int	result = -1;
 		char	   *localeptr;
 
-		if (default_locale.provider == COLLPROVIDER_ICU)
+		if (default_locale->provider == COLLPROVIDER_ICU)
 			return false;
 
 		if (result >= 0)
@@ -1386,7 +1392,7 @@ lc_ctype_is_c(Oid collation)
 		static int	result = -1;
 		char	   *localeptr;
 
-		if (default_locale.provider == COLLPROVIDER_ICU)
+		if (default_locale->provider == COLLPROVIDER_ICU)
 			return false;
 
 		if (result >= 0)
@@ -1417,38 +1423,6 @@ lc_ctype_is_c(Oid collation)
 	return (lookup_collation_cache(collation, true))->ctype_is_c;
 }
 
-struct pg_locale_struct default_locale;
-
-void
-make_icu_collator(const char *iculocstr,
-				  struct pg_locale_struct *resultp)
-{
-#ifdef USE_ICU
-	UCollator  *collator;
-	UErrorCode	status;
-
-	status = U_ZERO_ERROR;
-	collator = ucol_open(iculocstr, &status);
-	if (U_FAILURE(status))
-		ereport(ERROR,
-				(errmsg("could not open collator for locale \"%s\": %s",
-						iculocstr, u_errorName(status))));
-
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
-		icu_set_collation_attributes(collator, iculocstr);
-
-	/* We will leak this string if the caller errors later :-( */
-	resultp->info.icu.locale = MemoryContextStrdup(TopMemoryContext, iculocstr);
-	resultp->info.icu.ucol = collator;
-#else							/* not USE_ICU */
-	/* could get here if a collation was created by a build with ICU */
-	ereport(ERROR,
-			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
-			 errmsg("ICU is not supported in this build")));
-#endif							/* not USE_ICU */
-}
-
-
 /* simple subroutine for reporting errors from newlocale() */
 #ifdef HAVE_LOCALE_T
 static void
@@ -1482,6 +1456,261 @@ report_newlocale_failure(const char *localename)
 #endif							/* HAVE_LOCALE_T */
 
 
+/*
+ * Construct a new pg_locale_t object.
+ *
+ * Passing NULL for the version is allowed; and even if it is specified, the
+ * result may or may not have an exactly matching version. Other parameters
+ * are required. Caller should pass isdefault=true if initializing
+ * default_locale; false otherwise.
+ *
+ * Structures are allocated in CurrentMemoryContext. The libc locale_t or
+ * UCollator is not allocated in any memory context, so the caller should be
+ * sure to call pg_freelocale() to close it.
+ */
+static pg_locale_t
+pg_newlocale(char provider, bool isdefault, bool deterministic,
+			 const char *collate, const char *ctype, const char *version)
+{
+	pg_locale_t result = palloc0(sizeof(struct pg_locale_struct));
+
+	/*
+	 * If COLLPROVIDER_DEFAULT, caller should use default_locale or NULL
+	 * instead.
+	 */
+	Assert(provider != COLLPROVIDER_DEFAULT);
+
+	if (provider == COLLPROVIDER_LIBC && isdefault)
+	{
+		/*
+		 * When the default locale is libc, the actual locale settings are
+		 * controlled by setlocale(), so there's nothing to do here.
+		 */
+	}
+	else if (provider == COLLPROVIDER_LIBC)
+	{
+#ifdef HAVE_LOCALE_T
+		locale_t        loc;
+
+		/* newlocale's result may be leaked if we encounter an error */
+
+		if (strcmp(collate, ctype) == 0)
+		{
+			/* Normal case where they're the same */
+			errno = 0;
+#ifndef WIN32
+			loc = newlocale(LC_COLLATE_MASK | LC_CTYPE_MASK, collate,
+							NULL);
+#else
+			loc = _create_locale(LC_ALL, collate);
+#endif
+			if (!loc)
+				report_newlocale_failure(collate);
+		}
+		else
+		{
+#ifndef WIN32
+			/* We need two newlocale() steps */
+			locale_t	loc1;
+
+			errno = 0;
+			loc1 = newlocale(LC_COLLATE_MASK, collate, NULL);
+			if (!loc1)
+				report_newlocale_failure(collate);
+			errno = 0;
+			loc = newlocale(LC_CTYPE_MASK, ctype, loc1);
+			if (!loc)
+				report_newlocale_failure(ctype);
+#else
+
+			/*
+			 * XXX The _create_locale() API doesn't appear to support
+			 * this. Could perhaps be worked around by changing
+			 * pg_locale_t to contain two separate fields.
+			 */
+			ereport(ERROR,
+					(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
+					 errmsg("collations with different collate and ctype values are not supported on this platform")));
+#endif
+		}
+
+		result->info.libc.lt = loc;
+#else							/* not HAVE_LOCALE_T */
+		/* platform that doesn't support locale_t */
+		ereport(ERROR,
+				(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
+				 errmsg("collation provider LIBC is not supported on this platform")));
+#endif							/* not HAVE_LOCALE_T */
+	}
+#ifdef USE_ICU
+	else if (provider == COLLPROVIDER_ICU)
+	{
+		UCollator  *collator;
+		UErrorCode	status;
+
+		/* collator may be leaked if we encounter an error */
+
+		status = U_ZERO_ERROR;
+		collator = ucol_open(collate, &status);
+		if (U_FAILURE(status))
+			ereport(ERROR,
+					(errmsg("could not open collator for locale \"%s\": %s",
+							collate, u_errorName(status))));
+
+		if (U_ICU_VERSION_MAJOR_NUM < 54)
+			icu_set_collation_attributes(collator, collate);
+
+		result->info.icu.ucol = collator;
+	}
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", provider);
+
+	result->provider = provider;
+	result->deterministic = deterministic;
+	result->collate = pstrdup(collate);
+	result->ctype = pstrdup(ctype);
+
+	return result;
+}
+
+/*
+ * Move resources from the given locale into the given mcxt, consuming and
+ * freeing the given locale and returning the new one.
+ */
+static pg_locale_t
+pg_movelocale(MemoryContext mcxt, pg_locale_t *plocale)
+{
+	pg_locale_t locale = *plocale;
+	pg_locale_t result;
+
+	Assert(locale != default_locale);
+
+	result = MemoryContextAllocZero(mcxt, sizeof(struct pg_locale_struct));
+
+	result->provider = locale->provider;
+	result->deterministic = locale->deterministic;
+
+	if (locale->provider == COLLPROVIDER_LIBC)
+	{
+#ifdef HAVE_LOCALE_T
+		if (locale->info.libc.lt != NULL)
+		{
+			/* not in a memory context; just reassign the pointer */
+			result->info.libc.lt = locale->info.libc.lt;
+		}
+#endif
+	}
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+	{
+		/* not in a memory context; just reassign the pointer */
+		result->info.icu.ucol = locale->info.icu.ucol;
+	}
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	result->collate = MemoryContextStrdup(mcxt, locale->collate);
+	pfree(locale->collate);
+
+	result->ctype = MemoryContextStrdup(mcxt, locale->ctype);
+	pfree(locale->ctype);
+
+	pfree(locale);
+	*plocale = NULL;
+
+	return result;
+}
+
+/*
+ * Free pg_locale_t and close libc locale_t or UCollator.
+ */
+#ifdef USE_ICU
+static void
+pg_freelocale(pg_locale_t locale)
+{
+	if (!locale)
+		return;
+
+	Assert(locale != default_locale);
+
+	pfree(locale->collate);
+	pfree(locale->ctype);
+
+	if (locale->provider == COLLPROVIDER_LIBC)
+	{
+#ifdef HAVE_LOCALE_T
+		if (locale->info.libc.lt != NULL)
+		{
+#ifndef WIN32
+			freelocale(locale->info.libc.lt);
+#else
+			_free_locale(locale->info.libc.lt);
+#endif
+		}
+#endif
+	}
+#ifdef USE_ICU
+	else if (locale->provider == COLLPROVIDER_ICU)
+	{
+		ucol_close(locale->info.icu.ucol);
+	}
+#endif
+	else
+		/* shouldn't happen */
+		elog(ERROR, "unsupported collprovider: %c", locale->provider);
+
+	pfree(locale);
+}
+#endif
+
+/*
+ * Accessor so that callers don't need to include pg_locale_internal.h.
+ */
+bool
+pg_locale_deterministic(pg_locale_t locale)
+{
+	if (locale == NULL)
+		return true;
+	else
+		return locale->deterministic;
+}
+
+/*
+ * Initialize default database locale.
+ */
+void
+init_default_locale(char provider, const char *collate, const char *ctype,
+					const char *version)
+{
+	pg_locale_t temp_locale;
+	bool deterministic;
+
+	/*
+	 * Default locale is currently always deterministic.  Nondeterministic
+	 * locales currently don't support pattern matching, which would break a
+	 * lot of things if applied globally.
+	 */
+	deterministic = true;
+	temp_locale = pg_newlocale(provider, true, deterministic, collate,
+							   ctype, version);
+
+	default_locale = pg_movelocale(TopMemoryContext, &temp_locale);
+}
+
+/*
+ * Return palloc'd version string for the default locale.
+ */
+char *
+default_locale_collation_version()
+{
+	return get_collation_actual_version(default_locale->provider,
+										default_locale->collate);
+}
+
 /*
  * Create a locale_t from a collation OID.  Results are cached for the
  * lifetime of the backend.  Thus, do not free the result with freelocale().
@@ -1506,8 +1735,8 @@ pg_newlocale_from_collation(Oid collid)
 
 	if (collid == DEFAULT_COLLATION_OID)
 	{
-		if (default_locale.provider == COLLPROVIDER_ICU)
-			return &default_locale;
+		if (default_locale->provider == COLLPROVIDER_ICU)
+			return default_locale;
 		else
 			return (pg_locale_t) 0;
 	}
@@ -1519,107 +1748,65 @@ pg_newlocale_from_collation(Oid collid)
 		/* We haven't computed this yet in this session, so do it */
 		HeapTuple	tp;
 		Form_pg_collation collform;
-		struct pg_locale_struct result;
-		pg_locale_t resultp;
+		pg_locale_t temp_locale;
+		pg_locale_t perm_locale;
 		Datum		datum;
 		bool		isnull;
+		char	   *collate;
+		char	   *ctype;
+		char	   *collversionstr;
 
 		tp = SearchSysCache1(COLLOID, ObjectIdGetDatum(collid));
 		if (!HeapTupleIsValid(tp))
 			elog(ERROR, "cache lookup failed for collation %u", collid);
 		collform = (Form_pg_collation) GETSTRUCT(tp);
 
-		/* We'll fill in the result struct locally before allocating memory */
-		memset(&result, 0, sizeof(result));
-		result.provider = collform->collprovider;
-		result.deterministic = collform->collisdeterministic;
+		datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collversion,
+								&isnull);
+		if (!isnull)
+			collversionstr = TextDatumGetCString(datum);
+		else
+			collversionstr = NULL;
 
 		if (collform->collprovider == COLLPROVIDER_LIBC)
 		{
-#ifdef HAVE_LOCALE_T
-			const char *collcollate;
-			const char *collctype pg_attribute_unused();
-			locale_t	loc;
-
-			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collcollate, &isnull);
+			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collcollate,
+									&isnull);
 			Assert(!isnull);
-			collcollate = TextDatumGetCString(datum);
-			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collctype, &isnull);
+			collate = TextDatumGetCString(datum);
+			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collctype,
+									&isnull);
 			Assert(!isnull);
-			collctype = TextDatumGetCString(datum);
-
-			if (strcmp(collcollate, collctype) == 0)
-			{
-				/* Normal case where they're the same */
-				errno = 0;
-#ifndef WIN32
-				loc = newlocale(LC_COLLATE_MASK | LC_CTYPE_MASK, collcollate,
-								NULL);
-#else
-				loc = _create_locale(LC_ALL, collcollate);
-#endif
-				if (!loc)
-					report_newlocale_failure(collcollate);
-			}
-			else
-			{
-#ifndef WIN32
-				/* We need two newlocale() steps */
-				locale_t	loc1;
-
-				errno = 0;
-				loc1 = newlocale(LC_COLLATE_MASK, collcollate, NULL);
-				if (!loc1)
-					report_newlocale_failure(collcollate);
-				errno = 0;
-				loc = newlocale(LC_CTYPE_MASK, collctype, loc1);
-				if (!loc)
-					report_newlocale_failure(collctype);
-#else
-
-				/*
-				 * XXX The _create_locale() API doesn't appear to support
-				 * this. Could perhaps be worked around by changing
-				 * pg_locale_t to contain two separate fields.
-				 */
-				ereport(ERROR,
-						(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
-						 errmsg("collations with different collate and ctype values are not supported on this platform")));
-#endif
-			}
-
-			result.info.lt = loc;
-#else							/* not HAVE_LOCALE_T */
-			/* platform that doesn't support locale_t */
-			ereport(ERROR,
-					(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
-					 errmsg("collation provider LIBC is not supported on this platform")));
-#endif							/* not HAVE_LOCALE_T */
+			ctype = TextDatumGetCString(datum);
 		}
+#ifdef USE_ICU
 		else if (collform->collprovider == COLLPROVIDER_ICU)
 		{
-			const char *iculocstr;
-
-			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_colliculocale, &isnull);
+			datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_colliculocale,
+									&isnull);
 			Assert(!isnull);
-			iculocstr = TextDatumGetCString(datum);
-			make_icu_collator(iculocstr, &result);
+			collate = TextDatumGetCString(datum);
+
+			/* for ICU, collate and ctype are both set from iculocale */
+			ctype = collate;
 		}
+#endif
+		else
+			/* shouldn't happen */
+			elog(ERROR, "unsupported collprovider: %c", collform->collprovider);
 
-		datum = SysCacheGetAttr(COLLOID, tp, Anum_pg_collation_collversion,
-								&isnull);
-		if (!isnull)
+		temp_locale = pg_newlocale(collform->collprovider, false,
+								   collform->collisdeterministic,
+								   collate, ctype, collversionstr);
+
+		ReleaseSysCache(tp);
+
+		if (collversionstr != NULL)
 		{
 			char	   *actual_versionstr;
-			char	   *collversionstr;
-
-			collversionstr = TextDatumGetCString(datum);
 
-			datum = SysCacheGetAttr(COLLOID, tp, collform->collprovider == COLLPROVIDER_ICU ? Anum_pg_collation_colliculocale : Anum_pg_collation_collcollate, &isnull);
-			Assert(!isnull);
+			actual_versionstr = get_collation_actual_version(collform->collprovider, collate);
 
-			actual_versionstr = get_collation_actual_version(collform->collprovider,
-															 TextDatumGetCString(datum));
 			if (!actual_versionstr)
 			{
 				/*
@@ -1646,13 +1833,10 @@ pg_newlocale_from_collation(Oid collid)
 															NameStr(collform->collname)))));
 		}
 
-		ReleaseSysCache(tp);
-
-		/* We'll keep the pg_locale_t structures in TopMemoryContext */
-		resultp = MemoryContextAlloc(TopMemoryContext, sizeof(*resultp));
-		*resultp = result;
+		/* move into TopMemoryContext */
+		perm_locale = pg_movelocale(TopMemoryContext, &temp_locale);
 
-		cache_entry->locale = resultp;
+		cache_entry->locale = perm_locale;
 	}
 
 	return cache_entry->locale;
@@ -1812,7 +1996,7 @@ pg_strncoll_libc_win32_utf8(const char *arg1, size_t len1, const char *arg2,
 	errno = 0;
 #ifdef HAVE_LOCALE_T
 	if (locale)
-		result = wcscoll_l((LPWSTR) a1p, (LPWSTR) a2p, locale->info.lt);
+		result = wcscoll_l((LPWSTR) a1p, (LPWSTR) a2p, locale->info.libc.lt);
 	else
 #endif
 		result = wcscoll((LPWSTR) a1p, (LPWSTR) a2p);
@@ -1855,7 +2039,7 @@ pg_strcoll_libc(const char *arg1, const char *arg2, pg_locale_t locale)
 	if (locale)
 	{
 #ifdef HAVE_LOCALE_T
-		result = strcoll_l(arg1, arg2, locale->info.lt);
+		result = strcoll_l(arg1, arg2, locale->info.libc.lt);
 #else
 		/* shouldn't happen */
 		elog(ERROR, "unsupported collprovider: %c", locale->provider);
@@ -2108,7 +2292,7 @@ pg_strxfrm_libc(char *dest, const char *src, size_t destsize,
 #ifdef TRUST_STXFRM
 #ifdef HAVE_LOCALE_T
 	if (locale)
-		return strxfrm_l(dest, src, destsize, locale->info.lt);
+		return strxfrm_l(dest, src, destsize, locale->info.libc.lt);
 	else
 #endif
 		return strxfrm(dest, src, destsize);
@@ -2718,19 +2902,16 @@ void
 check_icu_locale(const char *icu_locale)
 {
 #ifdef USE_ICU
-	UCollator  *collator;
-	UErrorCode	status;
+	pg_locale_t locale;
 
-	status = U_ZERO_ERROR;
-	collator = ucol_open(icu_locale, &status);
-	if (U_FAILURE(status))
-		ereport(ERROR,
-				(errmsg("could not open collator for locale \"%s\": %s",
-						icu_locale, u_errorName(status))));
-
-	if (U_ICU_VERSION_MAJOR_NUM < 54)
-		icu_set_collation_attributes(collator, icu_locale);
-	ucol_close(collator);
+	/*
+	 * Whether it's deterministic doesn't matter in this case, because it
+	 * doesn't affect whether the locale is valid or not; and we're going to
+	 * discard the locale anyway.
+	 */
+	locale = pg_newlocale(COLLPROVIDER_ICU, false, true, icu_locale,
+						  icu_locale, NULL);
+	pg_freelocale(locale);
 #else
 	ereport(ERROR,
 			(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
@@ -2793,10 +2974,10 @@ wchar2char(char *to, const wchar_t *from, size_t tolen, pg_locale_t locale)
 #ifdef HAVE_LOCALE_T
 #ifdef HAVE_WCSTOMBS_L
 		/* Use wcstombs_l for nondefault locales */
-		result = wcstombs_l(to, from, tolen, locale->info.lt);
+		result = wcstombs_l(to, from, tolen, locale->info.libc.lt);
 #else							/* !HAVE_WCSTOMBS_L */
 		/* We have to temporarily set the locale as current ... ugh */
-		locale_t	save_locale = uselocale(locale->info.lt);
+		locale_t	save_locale = uselocale(locale->info.libc.lt);
 
 		result = wcstombs(to, from, tolen);
 
@@ -2870,10 +3051,10 @@ char2wchar(wchar_t *to, size_t tolen, const char *from, size_t fromlen,
 #ifdef HAVE_LOCALE_T
 #ifdef HAVE_MBSTOWCS_L
 			/* Use mbstowcs_l for nondefault locales */
-			result = mbstowcs_l(to, str, tolen, locale->info.lt);
+			result = mbstowcs_l(to, str, tolen, locale->info.libc.lt);
 #else							/* !HAVE_MBSTOWCS_L */
 			/* We have to temporarily set the locale as current ... ugh */
-			locale_t	save_locale = uselocale(locale->info.lt);
+			locale_t	save_locale = uselocale(locale->info.libc.lt);
 
 			result = mbstowcs(to, str, tolen);
 
diff --git a/src/backend/utils/adt/varchar.c b/src/backend/utils/adt/varchar.c
index d0bc528e9f..52f27d483d 100644
--- a/src/backend/utils/adt/varchar.c
+++ b/src/backend/utils/adt/varchar.c
@@ -757,7 +757,7 @@ bpchareq(PG_FUNCTION_ARGS)
 	else
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (locale_is_c || !mylocale || mylocale->deterministic)
+	if (locale_is_c || pg_locale_deterministic(mylocale))
 	{
 		/*
 		 * Since we only care about equality or not-equality, we can avoid all
@@ -802,7 +802,7 @@ bpcharne(PG_FUNCTION_ARGS)
 	else
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (locale_is_c || !mylocale || mylocale->deterministic)
+	if (locale_is_c || pg_locale_deterministic(mylocale))
 	{
 		/*
 		 * Since we only care about equality or not-equality, we can avoid all
@@ -1010,33 +1010,25 @@ hashbpchar(PG_FUNCTION_ARGS)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (!mylocale || mylocale->deterministic)
+	if (pg_locale_deterministic(mylocale))
 	{
 		result = hash_any((unsigned char *) keydata, keylen);
 	}
 	else
 	{
-#ifdef USE_ICU
-		if (mylocale->provider == COLLPROVIDER_ICU)
-		{
-			Size		bsize, rsize;
-			char	   *buf;
+		Size		bsize, rsize;
+		char	   *buf;
 
-			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
-			buf = palloc(bsize);
+		bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
+		buf = palloc(bsize);
 
-			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
-			if (rsize != bsize)
-				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+		rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+		if (rsize != bsize)
+			elog(ERROR, "pg_strnxfrm() returned unexpected result");
 
-			result = hash_any((uint8_t *) buf, bsize);
+		result = hash_any((uint8_t *) buf, bsize);
 
-			pfree(buf);
-		}
-		else
-#endif
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
+		pfree(buf);
 	}
 
 	/* Avoid leaking memory for toasted inputs */
@@ -1067,35 +1059,27 @@ hashbpcharextended(PG_FUNCTION_ARGS)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (!mylocale || mylocale->deterministic)
+	if (pg_locale_deterministic(mylocale))
 	{
 		result = hash_any_extended((unsigned char *) keydata, keylen,
 								   PG_GETARG_INT64(1));
 	}
 	else
 	{
-#ifdef USE_ICU
-		if (mylocale->provider == COLLPROVIDER_ICU)
-		{
-			Size		bsize, rsize;
-			char	   *buf;
+		Size		bsize, rsize;
+		char	   *buf;
 
-			bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
-			buf = palloc(bsize);
+		bsize = pg_strnxfrm(NULL, 0, keydata, keylen, mylocale);
+		buf = palloc(bsize);
 
-			rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
-			if (rsize != bsize)
-				elog(ERROR, "pg_strnxfrm() returned unexpected result");
+		rsize = pg_strnxfrm(buf, bsize, keydata, keylen, mylocale);
+		if (rsize != bsize)
+			elog(ERROR, "pg_strnxfrm() returned unexpected result");
 
-			result = hash_any_extended((uint8_t *) buf, bsize,
-									   PG_GETARG_INT64(1));
+		result = hash_any_extended((uint8_t *) buf, bsize,
+								   PG_GETARG_INT64(1));
 
-			pfree(buf);
-		}
-		else
-#endif
-			/* shouldn't happen */
-			elog(ERROR, "unsupported collprovider: %c", mylocale->provider);
+		pfree(buf);
 	}
 
 	PG_FREE_IF_COPY(key, 0);
diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c
index 2dfba4b488..a7c39d7afa 100644
--- a/src/backend/utils/adt/varlena.c
+++ b/src/backend/utils/adt/varlena.c
@@ -1203,7 +1203,7 @@ text_position_setup(text *t1, text *t2, Oid collid, TextPositionState *state)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (mylocale && !mylocale->deterministic)
+	if (!pg_locale_deterministic(mylocale))
 		ereport(ERROR,
 				(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
 				 errmsg("nondeterministic collations are not supported for substring searches")));
@@ -1601,7 +1601,7 @@ texteq(PG_FUNCTION_ARGS)
 	else
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (locale_is_c || !mylocale || mylocale->deterministic)
+	if (locale_is_c || pg_locale_deterministic(mylocale))
 	{
 		Datum		arg1 = PG_GETARG_DATUM(0);
 		Datum		arg2 = PG_GETARG_DATUM(1);
@@ -1660,7 +1660,7 @@ textne(PG_FUNCTION_ARGS)
 	else
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (locale_is_c || !mylocale || mylocale->deterministic)
+	if (locale_is_c || pg_locale_deterministic(mylocale))
 	{
 		Datum		arg1 = PG_GETARG_DATUM(0);
 		Datum		arg2 = PG_GETARG_DATUM(1);
@@ -1774,7 +1774,7 @@ text_starts_with(PG_FUNCTION_ARGS)
 	if (!lc_collate_is_c(collid))
 		mylocale = pg_newlocale_from_collation(collid);
 
-	if (mylocale && !mylocale->deterministic)
+	if (!pg_locale_deterministic(mylocale))
 		ereport(ERROR,
 				(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
 				 errmsg("nondeterministic collations are not supported for substring searches")));
diff --git a/src/backend/utils/init/postinit.c b/src/backend/utils/init/postinit.c
index a990c833c5..c5528cbf64 100644
--- a/src/backend/utils/init/postinit.c
+++ b/src/backend/utils/init/postinit.c
@@ -317,6 +317,7 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 	char	   *collate;
 	char	   *ctype;
 	char	   *iculocale;
+	char	   *collversionstr;
 
 	/* Fetch our pg_database row normally, via syscache */
 	tup = SearchSysCache1(DATABASEOID, ObjectIdGetDatum(MyDatabaseId));
@@ -424,35 +425,32 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 		datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_daticulocale, &isnull);
 		Assert(!isnull);
 		iculocale = TextDatumGetCString(datum);
-		make_icu_collator(iculocale, &default_locale);
 	}
 	else
 		iculocale = NULL;
 
-	default_locale.provider = dbform->datlocprovider;
+	datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_datcollversion,
+							&isnull);
+	if (!isnull)
+		collversionstr = TextDatumGetCString(datum);
+	else
+		collversionstr = NULL;
 
-	/*
-	 * Default locale is currently always deterministic.  Nondeterministic
-	 * locales currently don't support pattern matching, which would break a
-	 * lot of things if applied globally.
-	 */
-	default_locale.deterministic = true;
+	init_default_locale(dbform->datlocprovider,
+						dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : collate,
+						dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : ctype,
+						collversionstr);
 
 	/*
 	 * Check collation version.  See similar code in
 	 * pg_newlocale_from_collation().  Note that here we warn instead of error
 	 * in any case, so that we don't prevent connecting.
 	 */
-	datum = SysCacheGetAttr(DATABASEOID, tup, Anum_pg_database_datcollversion,
-							&isnull);
-	if (!isnull)
+	if (collversionstr != NULL)
 	{
 		char	   *actual_versionstr;
-		char	   *collversionstr;
 
-		collversionstr = TextDatumGetCString(datum);
-
-		actual_versionstr = get_collation_actual_version(dbform->datlocprovider, dbform->datlocprovider == COLLPROVIDER_ICU ? iculocale : collate);
+		actual_versionstr = default_locale_collation_version();
 		if (!actual_versionstr)
 			/* should not happen */
 			elog(WARNING,
@@ -470,6 +468,8 @@ CheckMyDatabase(const char *name, bool am_superuser, bool override_allow_connect
 							 "or build PostgreSQL with the right library version.",
 							 quote_identifier(name))));
 	}
+	else
+		collversionstr = NULL;
 
 	/* Make the locale settings visible as GUC variables, too */
 	SetConfigOption("lc_collate", collate, PGC_INTERNAL, PGC_S_DYNAMIC_DEFAULT);
diff --git a/src/include/utils/pg_locale.h b/src/include/utils/pg_locale.h
index ceab0d4307..0d7bc0534f 100644
--- a/src/include/utils/pg_locale.h
+++ b/src/include/utils/pg_locale.h
@@ -15,22 +15,6 @@
 #if defined(LOCALE_T_IN_XLOCALE) || defined(WCSTOMBS_L_IN_XLOCALE)
 #include <xlocale.h>
 #endif
-#ifdef USE_ICU
-#include <unicode/ucol.h>
-#endif
-
-#ifdef USE_ICU
-/*
- * ucol_strcollUTF8() was introduced in ICU 50, but it is buggy before ICU 53.
- * (see
- * <https://www.postgresql.org/message-id/flat/f1438ec6-22aa-4029-9a3b-26f79d330e72%40manitou-mail.org>)
- */
-#if U_ICU_VERSION_MAJOR_NUM >= 53
-#define HAVE_UCOL_STRCOLLUTF8 1
-#else
-#undef HAVE_UCOL_STRCOLLUTF8
-#endif
-#endif
 
 /* use for libc locale names */
 #define LOCALE_NAME_BUFLEN 128
@@ -64,39 +48,12 @@ extern struct lconv *PGLC_localeconv(void);
 extern void cache_locale_time(void);
 
 
-/*
- * We define our own wrapper around locale_t so we can keep the same
- * function signatures for all builds, while not having to create a
- * fake version of the standard type locale_t in the global namespace.
- * pg_locale_t is occasionally checked for truth, so make it a pointer.
- */
-struct pg_locale_struct
-{
-	char		provider;
-	bool		deterministic;
-	union
-	{
-#ifdef HAVE_LOCALE_T
-		locale_t	lt;
-#endif
-#ifdef USE_ICU
-		struct
-		{
-			const char *locale;
-			UCollator  *ucol;
-		}			icu;
-#endif
-		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
-	}			info;
-};
-
 typedef struct pg_locale_struct *pg_locale_t;
 
-extern PGDLLIMPORT struct pg_locale_struct default_locale;
-
-extern void make_icu_collator(const char *iculocstr,
-							  struct pg_locale_struct *resultp);
-
+extern void init_default_locale(char provider, const char *collate,
+								const char *ctype, const char *version);
+extern char *default_locale_collation_version(void);
+extern bool pg_locale_deterministic(pg_locale_t locale);
 extern pg_locale_t pg_newlocale_from_collation(Oid collid);
 
 extern char *get_collation_actual_version(char collprovider, const char *collcollate);
@@ -114,10 +71,6 @@ extern size_t pg_strxfrm_prefix(char *dest, const char *src, size_t destsize,
 extern size_t pg_strnxfrm_prefix(char *dest, size_t destsize, const char *src,
 								 size_t srclen, pg_locale_t locale);
 
-#ifdef USE_ICU
-extern int32_t icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes);
-extern int32_t icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar);
-#endif
 extern void check_icu_locale(const char *icu_locale);
 
 /* These functions convert from/to libc's wchar_t, *not* pg_wchar_t */
diff --git a/src/include/utils/pg_locale_internal.h b/src/include/utils/pg_locale_internal.h
new file mode 100644
index 0000000000..33465ad92d
--- /dev/null
+++ b/src/include/utils/pg_locale_internal.h
@@ -0,0 +1,68 @@
+/*-----------------------------------------------------------------------
+ *
+ * PostgreSQL locale utilities
+ *
+ * src/include/utils/pg_locale_internal.h
+ *
+ * Copyright (c) 2002-2022, PostgreSQL Global Development Group
+ *
+ *-----------------------------------------------------------------------
+ */
+
+
+#ifndef _PG_LOCALE_INTERNAL_
+#define _PG_LOCALE_INTERNAL_
+
+#ifdef USE_ICU
+#include <unicode/ucol.h>
+#endif
+
+#ifdef USE_ICU
+/*
+ * ucol_strcollUTF8() was introduced in ICU 50, but it is buggy before ICU 53.
+ * (see
+ * <https://www.postgresql.org/message-id/flat/f1438ec6-22aa-4029-9a3b-26f79d330e72%40manitou-mail.org>)
+ */
+#if U_ICU_VERSION_MAJOR_NUM >= 53
+#define HAVE_UCOL_STRCOLLUTF8 1
+#else
+#undef HAVE_UCOL_STRCOLLUTF8
+#endif
+#endif
+
+/*
+ * We define our own wrapper around locale_t so we can keep the same
+ * function signatures for all builds, while not having to create a
+ * fake version of the standard type locale_t in the global namespace.
+ * pg_locale_t is occasionally checked for truth, so make it a pointer.
+ */
+struct pg_locale_struct
+{
+	char		provider;
+	bool		deterministic;
+	char	   *collate;
+	char	   *ctype;
+	union
+	{
+#ifdef HAVE_LOCALE_T
+		struct
+		{
+			locale_t	lt;
+		}			libc;
+#endif
+#ifdef USE_ICU
+		struct
+		{
+			UCollator	*ucol;
+		}			icu;
+#endif
+		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
+	}			info;
+};
+
+#ifdef USE_ICU
+extern int32_t icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes);
+extern int32_t icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar);
+#endif
+
+#endif							/* _PG_LOCALE_INTERNAL_ */
-- 
2.34.1



  [text/x-patch] v2-0004-Add-method-structure-toward-ICU-multi-library-sup.patch (28.8K, ../../[email protected]/5-v2-0004-Add-method-structure-toward-ICU-multi-library-sup.patch)
  download | inline diff:
From daef4c5f6ebf6cd73dcae5402feb5ec269a50650 Mon Sep 17 00:00:00 2001
From: Jeff Davis <[email protected]>
Date: Wed, 7 Dec 2022 11:07:31 -0800
Subject: [PATCH v2 4/6] Add method structure, toward ICU multi-library
 support.

Introduce structure pg_icu_library, which holds pointers to each
required ICU method, and store this as part of pg_locale_t. Each call
to an ICU function instead goes through this structure, so that it can
more easily be replaced by a non-builtin ICU library.

This is a step toward support for multiple ICU libraries, to allow
control over precisely which version of a library is used and prevent
problems from subtle changes in collation order.

Author: Thomas Munro, Jeff Davis
---
 src/backend/commands/collationcmds.c   |  17 +-
 src/backend/utils/adt/formatting.c     |  67 ++++++--
 src/backend/utils/adt/pg_locale.c      | 227 +++++++++++++++++--------
 src/include/utils/pg_locale_internal.h | 113 +++++++++++-
 4 files changed, 322 insertions(+), 102 deletions(-)

diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index 9e84da4891..16b1bcbdc0 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -560,13 +560,14 @@ get_icu_language_tag(const char *localename)
 {
 	char		buf[ULOC_FULLNAME_CAPACITY];
 	UErrorCode	status;
+	pg_icu_library *iculib = get_builtin_icu_library();
 
 	status = U_ZERO_ERROR;
-	uloc_toLanguageTag(localename, buf, sizeof(buf), true, &status);
+	iculib->toLanguageTag(localename, buf, sizeof(buf), true, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not convert locale name \"%s\" to language tag: %s",
-						localename, u_errorName(status))));
+						localename, iculib->errorName(status))));
 
 	return pstrdup(buf);
 }
@@ -585,11 +586,12 @@ get_icu_locale_comment(const char *localename)
 	int32		len_uchar;
 	int32		i;
 	char	   *result;
+	pg_icu_library *iculib = get_builtin_icu_library();
 
 	status = U_ZERO_ERROR;
-	len_uchar = uloc_getDisplayName(localename, "en",
-									displayname, lengthof(displayname),
-									&status);
+	len_uchar = iculib->getDisplayName(localename, "en",
+									   displayname, lengthof(displayname),
+									   &status);
 	if (U_FAILURE(status))
 		return NULL;			/* no good reason to raise an error */
 
@@ -809,12 +811,13 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 	{
 		int			i;
+		pg_icu_library *iculib = get_builtin_icu_library();
 
 		/*
 		 * Start the loop at -1 to sneak in the root locale without too much
 		 * code duplication.
 		 */
-		for (i = -1; i < uloc_countAvailable(); i++)
+		for (i = -1; i < iculib->countAvailable(); i++)
 		{
 			const char *name;
 			char	   *langtag;
@@ -825,7 +828,7 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 			if (i == -1)
 				name = "";		/* ICU root locale */
 			else
-				name = uloc_getAvailable(i);
+				name = iculib->getAvailable(i);
 
 			langtag = get_icu_language_tag(name);
 			iculocstr = U_ICU_VERSION_MAJOR_NUM >= 54 ? langtag : name;
diff --git a/src/backend/utils/adt/formatting.c b/src/backend/utils/adt/formatting.c
index a4bc7fa5f5..289aa569de 100644
--- a/src/backend/utils/adt/formatting.c
+++ b/src/backend/utils/adt/formatting.c
@@ -1600,6 +1600,11 @@ typedef int32_t (*ICU_Convert_Func) (UChar *dest, int32_t destCapacity,
 									 const UChar *src, int32_t srcLength,
 									 const char *locale,
 									 UErrorCode *pErrorCode);
+typedef int32_t (*ICU_Convert_BI_Func) (UChar *dest, int32_t destCapacity,
+										const UChar *src, int32_t srcLength,
+										UBreakIterator *bi,
+										const char *locale,
+										UErrorCode *pErrorCode);
 
 static int32_t
 icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
@@ -1607,6 +1612,7 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 {
 	UErrorCode	status;
 	int32_t		len_dest;
+	pg_icu_library *iculib = PG_ICU_LIB(mylocale);
 
 	len_dest = len_source;		/* try first with same length */
 	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
@@ -1624,18 +1630,42 @@ icu_convert_case(ICU_Convert_Func func, pg_locale_t mylocale,
 	}
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("case conversion failed: %s", u_errorName(status))));
+				(errmsg("case conversion failed: %s",
+						iculib->errorName(status))));
 	return len_dest;
 }
 
+/*
+ * Like icu_convert_case, but func takes a break iterator (which we don't
+ * make use of).
+ */
 static int32_t
-u_strToTitle_default_BI(UChar *dest, int32_t destCapacity,
-						const UChar *src, int32_t srcLength,
-						const char *locale,
-						UErrorCode *pErrorCode)
+icu_convert_case_bi(ICU_Convert_BI_Func func, pg_locale_t mylocale,
+					UChar **buff_dest, UChar *buff_source, int32_t len_source)
 {
-	return u_strToTitle(dest, destCapacity, src, srcLength,
-						NULL, locale, pErrorCode);
+	UErrorCode	status;
+	int32_t		len_dest;
+	pg_icu_library *iculib = PG_ICU_LIB(mylocale);
+
+	len_dest = len_source;		/* try first with same length */
+	*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+	status = U_ZERO_ERROR;
+	len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+					mylocale->ctype, &status);
+	if (status == U_BUFFER_OVERFLOW_ERROR)
+	{
+		/* try again with adjusted length */
+		pfree(*buff_dest);
+		*buff_dest = palloc(len_dest * sizeof(**buff_dest));
+		status = U_ZERO_ERROR;
+		len_dest = func(*buff_dest, len_dest, buff_source, len_source, NULL,
+						mylocale->ctype, &status);
+	}
+	if (U_FAILURE(status))
+		ereport(ERROR,
+				(errmsg("case conversion failed: %s",
+						iculib->errorName(status))));
+	return len_dest;
 }
 
 #endif							/* USE_ICU */
@@ -1701,11 +1731,12 @@ str_tolower(const char *buff, size_t nbytes, Oid collid)
 			int32_t		len_conv;
 			UChar	   *buff_uchar;
 			UChar	   *buff_conv;
+			pg_icu_library *iculib = PG_ICU_LIB(mylocale);
 
-			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToLower, mylocale,
+			len_uchar = icu_to_uchar(iculib, &buff_uchar, buff, nbytes);
+			len_conv = icu_convert_case(iculib->strToLower, mylocale,
 										&buff_conv, buff_uchar, len_uchar);
-			icu_from_uchar(&result, buff_conv, len_conv);
+			icu_from_uchar(iculib, &result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
 		}
@@ -1823,11 +1854,12 @@ str_toupper(const char *buff, size_t nbytes, Oid collid)
 						len_conv;
 			UChar	   *buff_uchar;
 			UChar	   *buff_conv;
+			pg_icu_library *iculib = PG_ICU_LIB(mylocale);
 
-			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToUpper, mylocale,
+			len_uchar = icu_to_uchar(iculib, &buff_uchar, buff, nbytes);
+			len_conv = icu_convert_case(iculib->strToUpper, mylocale,
 										&buff_conv, buff_uchar, len_uchar);
-			icu_from_uchar(&result, buff_conv, len_conv);
+			icu_from_uchar(iculib, &result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
 		}
@@ -1946,11 +1978,12 @@ str_initcap(const char *buff, size_t nbytes, Oid collid)
 						len_conv;
 			UChar	   *buff_uchar;
 			UChar	   *buff_conv;
+			pg_icu_library *iculib = PG_ICU_LIB(mylocale);
 
-			len_uchar = icu_to_uchar(&buff_uchar, buff, nbytes);
-			len_conv = icu_convert_case(u_strToTitle_default_BI, mylocale,
-										&buff_conv, buff_uchar, len_uchar);
-			icu_from_uchar(&result, buff_conv, len_conv);
+			len_uchar = icu_to_uchar(iculib, &buff_uchar, buff, nbytes);
+			len_conv = icu_convert_case_bi(iculib->strToTitle, mylocale,
+										   &buff_conv, buff_uchar, len_uchar);
+			icu_from_uchar(iculib, &result, buff_conv, len_conv);
 			pfree(buff_uchar);
 			pfree(buff_conv);
 		}
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 0a19845df4..4daff8b7b5 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -70,6 +70,8 @@
 
 #ifdef USE_ICU
 #include <unicode/ucnv.h>
+#include <unicode/ulocdata.h>
+#include <unicode/ustring.h>
 #endif
 
 #ifdef __GLIBC__
@@ -135,6 +137,7 @@ static char *IsoLocaleName(const char *);
 static pg_locale_t default_locale = NULL;
 
 #ifdef USE_ICU
+
 /*
  * Converter object for converting between ICU's UChar strings and C strings
  * in database encoding.  Since the database encoding doesn't change, we only
@@ -142,13 +145,17 @@ static pg_locale_t default_locale = NULL;
  */
 static UConverter *icu_converter = NULL;
 
-static void init_icu_converter(void);
-static size_t uchar_length(UConverter *converter,
+static void init_icu_converter(pg_icu_library *iculib);
+static size_t uchar_length(pg_icu_library *iculib,
+						   UConverter *converter,
 						   const char *str, size_t len);
-static int32_t uchar_convert(UConverter *converter,
+static int32_t uchar_convert(pg_icu_library *iculib,
+							 UConverter *converter,
 							 UChar *dest, int32_t destlen,
 							 const char *str, size_t srclen);
-static void icu_set_collation_attributes(UCollator *collator, const char *loc);
+static void icu_set_collation_attributes(pg_icu_library *iculib,
+										 UCollator *collator,
+										 const char *loc);
 #endif
 
 /*
@@ -1455,6 +1462,59 @@ report_newlocale_failure(const char *localename)
 }
 #endif							/* HAVE_LOCALE_T */
 
+#ifdef USE_ICU
+pg_icu_library *
+get_builtin_icu_library()
+{
+	pg_icu_library *lib;
+
+	/*
+	 * These assignments will fail to compile if an incompatible API change is
+	 * made to some future version of ICU, at which point we might need to
+	 * consider special treatment for different major version ranges, with
+	 * intermediate trampoline functions.
+	 */
+	lib = palloc0(sizeof(*lib));
+	lib->getICUVersion = u_getVersion;
+	lib->getUnicodeVersion = u_getUnicodeVersion;
+	lib->getCLDRVersion = ulocdata_getCLDRVersion;
+	lib->openCollator = ucol_open;
+	lib->closeCollator = ucol_close;
+	lib->getCollatorVersion = ucol_getVersion;
+	lib->getUCAVersion = ucol_getUCAVersion;
+	lib->versionToString = u_versionToString;
+	lib->strcoll = ucol_strcoll;
+	lib->strcollUTF8 = ucol_strcollUTF8;
+	lib->getSortKey = ucol_getSortKey;
+	lib->nextSortKeyPart = ucol_nextSortKeyPart;
+	lib->setUTF8 = uiter_setUTF8;
+	lib->errorName = u_errorName;
+	lib->strToUpper = u_strToUpper;
+	lib->strToLower = u_strToLower;
+	lib->strToTitle = u_strToTitle;
+	lib->setAttribute = ucol_setAttribute;
+	lib->openConverter = ucnv_open;
+	lib->closeConverter = ucnv_close;
+	lib->fromUChars = ucnv_fromUChars;
+	lib->toUChars = ucnv_toUChars;
+	lib->toLanguageTag = uloc_toLanguageTag;
+	lib->getDisplayName = uloc_getDisplayName;
+	lib->countAvailable = uloc_countAvailable;
+	lib->getAvailable = uloc_getAvailable;
+
+	/*
+	 * Also assert the size of a couple of types used as output buffers, as a
+	 * canary to tell us to add extra padding in the (unlikely) event that a
+	 * later release makes these values smaller.
+	 */
+	StaticAssertStmt(U_MAX_VERSION_STRING_LENGTH == 20,
+					 "u_versionToString output buffer size changed incompatibly");
+	StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
+					 "ucol_getVersion output buffer size changed incompatibly");
+
+	return lib;
+}
+#endif
 
 /*
  * Construct a new pg_locale_t object.
@@ -1547,20 +1607,22 @@ pg_newlocale(char provider, bool isdefault, bool deterministic,
 	{
 		UCollator  *collator;
 		UErrorCode	status;
+		pg_icu_library *iculib = get_builtin_icu_library();
 
 		/* collator may be leaked if we encounter an error */
 
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collate, &status);
+		collator = iculib->openCollator(collate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collate, u_errorName(status))));
+							collate, iculib->errorName(status))));
 
 		if (U_ICU_VERSION_MAJOR_NUM < 54)
-			icu_set_collation_attributes(collator, collate);
+			icu_set_collation_attributes(iculib, collator, collate);
 
 		result->info.icu.ucol = collator;
+		result->info.icu.lib = iculib;
 	}
 #endif
 	else
@@ -1607,6 +1669,10 @@ pg_movelocale(MemoryContext mcxt, pg_locale_t *plocale)
 	{
 		/* not in a memory context; just reassign the pointer */
 		result->info.icu.ucol = locale->info.icu.ucol;
+
+		result->info.icu.lib = MemoryContextAlloc(mcxt, sizeof(pg_icu_library));
+		memcpy(result->info.icu.lib, locale->info.icu.lib, sizeof(pg_icu_library));
+		pfree(locale->info.icu.lib);
 	}
 #endif
 	else
@@ -1656,7 +1722,9 @@ pg_freelocale(pg_locale_t locale)
 #ifdef USE_ICU
 	else if (locale->provider == COLLPROVIDER_ICU)
 	{
-		ucol_close(locale->info.icu.ucol);
+		pg_icu_library *iculib = PG_ICU_LIB(locale);
+		iculib->closeCollator(locale->info.icu.ucol);
+		pfree(locale->info.icu.lib);
 	}
 #endif
 	else
@@ -1858,17 +1926,18 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
+		pg_icu_library	*iculib = get_builtin_icu_library();
 
 		status = U_ZERO_ERROR;
-		collator = ucol_open(collcollate, &status);
+		collator = iculib->openCollator(collcollate, &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("could not open collator for locale \"%s\": %s",
-							collcollate, u_errorName(status))));
-		ucol_getVersion(collator, versioninfo);
-		ucol_close(collator);
+							collcollate, iculib->errorName(status))));
+		iculib->getCollatorVersion(collator, versioninfo);
+		iculib->closeCollator(collator);
 
-		u_versionToString(versioninfo, buf);
+		iculib->versionToString(versioninfo, buf);
 		collversion = pstrdup(buf);
 	}
 	else
@@ -2120,16 +2189,17 @@ pg_strncoll_icu_no_utf8(const char *arg1, size_t len1,
 	UChar	*uchar1,
 			*uchar2;
 	int		 result;
+	pg_icu_library *iculib = PG_ICU_LIB(locale);
 
 	Assert(locale->provider == COLLPROVIDER_ICU);
 #ifdef HAVE_UCOL_STRCOLLUTF8
 	Assert(GetDatabaseEncoding() != PG_UTF8);
 #endif
 
-	init_icu_converter();
+	init_icu_converter(iculib);
 
-	ulen1 = uchar_length(icu_converter, arg1, len1);
-	ulen2 = uchar_length(icu_converter, arg2, len2);
+	ulen1 = uchar_length(iculib, icu_converter, arg1, len1);
+	ulen2 = uchar_length(iculib, icu_converter, arg2, len2);
 
 	bufsize1 = (ulen1 + 1) * sizeof(UChar);
 	bufsize2 = (ulen2 + 1) * sizeof(UChar);
@@ -2140,12 +2210,12 @@ pg_strncoll_icu_no_utf8(const char *arg1, size_t len1,
 	uchar1 = (UChar *) buf;
 	uchar2 = (UChar *) (buf + bufsize1);
 
-	ulen1 = uchar_convert(icu_converter, uchar1, ulen1 + 1, arg1, len1);
-	ulen2 = uchar_convert(icu_converter, uchar2, ulen2 + 1, arg2, len2);
+	ulen1 = uchar_convert(iculib, icu_converter, uchar1, ulen1 + 1, arg1, len1);
+	ulen2 = uchar_convert(iculib, icu_converter, uchar2, ulen2 + 1, arg2, len2);
 
-	result = ucol_strcoll(locale->info.icu.ucol,
-						  uchar1, ulen1,
-						  uchar2, ulen2);
+	result = iculib->strcoll(locale->info.icu.ucol,
+							 uchar1, ulen1,
+							 uchar2, ulen2);
 
 	if (buf != sbuf)
 		pfree(buf);
@@ -2166,6 +2236,7 @@ pg_strncoll_icu(const char *arg1, size_t len1, const char *arg2, size_t len2,
 				pg_locale_t locale)
 {
 	int result;
+	pg_icu_library *iculib = PG_ICU_LIB(locale);
 
 	Assert(locale->provider == COLLPROVIDER_ICU);
 
@@ -2175,13 +2246,14 @@ pg_strncoll_icu(const char *arg1, size_t len1, const char *arg2, size_t len2,
 		UErrorCode	status;
 
 		status = U_ZERO_ERROR;
-		result = ucol_strcollUTF8(locale->info.icu.ucol,
-								  arg1, len1,
-								  arg2, len2,
-								  &status);
+		result = iculib->strcollUTF8(locale->info.icu.ucol,
+									 arg1, len1,
+									 arg2, len2,
+									 &status);
 		if (U_FAILURE(status))
 			ereport(ERROR,
-					(errmsg("collation failed: %s", u_errorName(status))));
+					(errmsg("collation failed: %s",
+							iculib->errorName(status))));
 	}
 	else
 #endif
@@ -2360,12 +2432,13 @@ pg_strnxfrm_icu(char *dest, const char *src, size_t srclen, size_t destsize,
 	int32_t	 ulen;
 	size_t   uchar_bsize;
 	Size	 result_bsize;
+	pg_icu_library *iculib = PG_ICU_LIB(locale);
 
 	Assert(locale->provider == COLLPROVIDER_ICU);
 
-	init_icu_converter();
+	init_icu_converter(iculib);
 
-	ulen = uchar_length(icu_converter, src, srclen);
+	ulen = uchar_length(iculib, icu_converter, src, srclen);
 
 	uchar_bsize = (ulen + 1) * sizeof(UChar);
 
@@ -2374,11 +2447,11 @@ pg_strnxfrm_icu(char *dest, const char *src, size_t srclen, size_t destsize,
 
 	uchar = (UChar *) buf;
 
-	ulen = uchar_convert(icu_converter, uchar, ulen + 1, src, srclen);
+	ulen = uchar_convert(iculib, icu_converter, uchar, ulen + 1, src, srclen);
 
-	result_bsize = ucol_getSortKey(locale->info.icu.ucol,
-								   uchar, ulen,
-								   (uint8_t *) dest, destsize);
+	result_bsize = iculib->getSortKey(locale->info.icu.ucol,
+									  uchar, ulen,
+									  (uint8_t *) dest, destsize);
 
 	if (buf != sbuf)
 		pfree(buf);
@@ -2407,13 +2480,14 @@ pg_strnxfrm_prefix_icu_no_utf8(char *dest, const char *src, size_t srclen,
 	UChar			*uchar = NULL;
 	size_t			 uchar_bsize;
 	Size			 result_bsize;
+	pg_icu_library	*iculib = PG_ICU_LIB(locale);
 
 	Assert(locale->provider == COLLPROVIDER_ICU);
 	Assert(GetDatabaseEncoding() != PG_UTF8);
 
-	init_icu_converter();
+	init_icu_converter(iculib);
 
-	ulen = uchar_length(icu_converter, src, srclen);
+	ulen = uchar_length(iculib, icu_converter, src, srclen);
 
 	uchar_bsize = (ulen + 1) * sizeof(UChar);
 
@@ -2422,21 +2496,19 @@ pg_strnxfrm_prefix_icu_no_utf8(char *dest, const char *src, size_t srclen,
 
 	uchar = (UChar *) buf;
 
-	ulen = uchar_convert(icu_converter, uchar, ulen + 1, src, srclen);
+	ulen = uchar_convert(iculib, icu_converter, uchar, ulen + 1, src, srclen);
 
 	uiter_setString(&iter, uchar, ulen);
 	state[0] = state[1] = 0;	/* won't need that again */
 	status = U_ZERO_ERROR;
-	result_bsize = ucol_nextSortKeyPart(locale->info.icu.ucol,
-										&iter,
-										state,
-										(uint8_t *) dest,
-										destsize,
-										&status);
+	result_bsize = iculib->nextSortKeyPart(
+		locale->info.icu.ucol, &iter, state,
+		(uint8_t *) dest, destsize, &status);
+
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("sort key generation failed: %s",
-						u_errorName(status))));
+						iculib->errorName(status))));
 
 	return result_bsize;
 }
@@ -2445,6 +2517,7 @@ static size_t
 pg_strnxfrm_prefix_icu(char *dest, const char *src, size_t srclen,
 					   size_t destsize, pg_locale_t locale)
 {
+	pg_icu_library *iculib = PG_ICU_LIB(locale);
 	size_t result;
 
 	Assert(locale->provider == COLLPROVIDER_ICU);
@@ -2455,19 +2528,17 @@ pg_strnxfrm_prefix_icu(char *dest, const char *src, size_t srclen,
 		uint32_t	state[2];
 		UErrorCode	status;
 
-		uiter_setUTF8(&iter, src, srclen);
+		iculib->setUTF8(&iter, src, srclen);
 		state[0] = state[1] = 0;	/* won't need that again */
 		status = U_ZERO_ERROR;
-		result = ucol_nextSortKeyPart(locale->info.icu.ucol,
-									  &iter,
-									  state,
-									  (uint8_t *) dest,
-									  destsize,
-									  &status);
+		result = iculib->nextSortKeyPart(
+			locale->info.icu.ucol, &iter, state,
+			(uint8_t *) dest, destsize, &status);
+
 		if (U_FAILURE(status))
 			ereport(ERROR,
 					(errmsg("sort key generation failed: %s",
-							u_errorName(status))));
+							iculib->errorName(status))));
 	}
 	else
 		result = pg_strnxfrm_prefix_icu_no_utf8(dest, src, srclen, destsize,
@@ -2667,7 +2738,7 @@ pg_strnxfrm_prefix(char *dest, size_t destsize, const char *src,
 
 #ifdef USE_ICU
 static void
-init_icu_converter(void)
+init_icu_converter(pg_icu_library *iculib)
 {
 	const char *icu_encoding_name;
 	UErrorCode	status;
@@ -2684,11 +2755,11 @@ init_icu_converter(void)
 						pg_encoding_to_char(GetDatabaseEncoding()))));
 
 	status = U_ZERO_ERROR;
-	conv = ucnv_open(icu_encoding_name, &status);
+	conv = iculib->openConverter(icu_encoding_name, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("could not open ICU converter for encoding \"%s\": %s",
-						icu_encoding_name, u_errorName(status))));
+						icu_encoding_name, iculib->errorName(status))));
 
 	icu_converter = conv;
 }
@@ -2697,14 +2768,15 @@ init_icu_converter(void)
  * Find length, in UChars, of given string if converted to UChar string.
  */
 static size_t
-uchar_length(UConverter *converter, const char *str, size_t len)
+uchar_length(pg_icu_library *iculib, UConverter *converter, const char *str, size_t len)
 {
 	UErrorCode	status = U_ZERO_ERROR;
 	int32_t		ulen;
-	ulen = ucnv_toUChars(converter, NULL, 0, str, len, &status);
+	ulen = iculib->toUChars(converter, NULL, 0, str, len, &status);
 	if (U_FAILURE(status) && status != U_BUFFER_OVERFLOW_ERROR)
 		ereport(ERROR,
-				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+				(errmsg("%s failed: %s", "ucnv_toUChars",
+						iculib->errorName(status))));
 	return ulen;
 }
 
@@ -2713,16 +2785,17 @@ uchar_length(UConverter *converter, const char *str, size_t len)
  * return the length (in UChars).
  */
 static int32_t
-uchar_convert(UConverter *converter, UChar *dest, int32_t destlen,
-			  const char *src, size_t srclen)
+uchar_convert(pg_icu_library *iculib, UConverter *converter, UChar *dest,
+			  int32_t destlen, const char *src, size_t srclen)
 {
 	UErrorCode	status = U_ZERO_ERROR;
 	int32_t		ulen;
 	status = U_ZERO_ERROR;
-	ulen = ucnv_toUChars(converter, dest, destlen, src, srclen, &status);
+	ulen = iculib->toUChars(converter, dest, destlen, src, srclen, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
-				(errmsg("%s failed: %s", "ucnv_toUChars", u_errorName(status))));
+				(errmsg("%s failed: %s", "ucnv_toUChars",
+						iculib->errorName(status))));
 	return ulen;
 }
 
@@ -2739,16 +2812,17 @@ uchar_convert(UConverter *converter, UChar *dest, int32_t destlen,
  * result length instead.
  */
 int32_t
-icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes)
+icu_to_uchar(pg_icu_library *iculib, UChar **buff_uchar, const char *buff,
+			 size_t nbytes)
 {
 	int32_t len_uchar;
 
-	init_icu_converter();
+	init_icu_converter(iculib);
 
-	len_uchar = uchar_length(icu_converter, buff, nbytes);
+	len_uchar = uchar_length(iculib, icu_converter, buff, nbytes);
 
 	*buff_uchar = palloc((len_uchar + 1) * sizeof(**buff_uchar));
-	len_uchar = uchar_convert(icu_converter,
+	len_uchar = uchar_convert(iculib, icu_converter,
 							  *buff_uchar, len_uchar + 1, buff, nbytes);
 
 	return len_uchar;
@@ -2766,30 +2840,32 @@ icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes)
  * The result string is nul-terminated.
  */
 int32_t
-icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar)
+icu_from_uchar(pg_icu_library *iculib, char **result, const UChar *buff_uchar,
+			   int32_t len_uchar)
 {
 	UErrorCode	status;
 	int32_t		len_result;
 
-	init_icu_converter();
+	init_icu_converter(iculib);
 
 	status = U_ZERO_ERROR;
-	len_result = ucnv_fromUChars(icu_converter, NULL, 0,
-								 buff_uchar, len_uchar, &status);
+	len_result = iculib->fromUChars(icu_converter, NULL, 0,
+									buff_uchar, len_uchar, &status);
 	if (U_FAILURE(status) && status != U_BUFFER_OVERFLOW_ERROR)
 		ereport(ERROR,
 				(errmsg("%s failed: %s", "ucnv_fromUChars",
-						u_errorName(status))));
+						iculib->errorName(status))));
 
 	*result = palloc(len_result + 1);
 
 	status = U_ZERO_ERROR;
-	len_result = ucnv_fromUChars(icu_converter, *result, len_result + 1,
-								 buff_uchar, len_uchar, &status);
+	len_result = iculib->fromUChars(icu_converter, *result,
+									len_result + 1, buff_uchar,
+									len_uchar, &status);
 	if (U_FAILURE(status))
 		ereport(ERROR,
 				(errmsg("%s failed: %s", "ucnv_fromUChars",
-						u_errorName(status))));
+						iculib->errorName(status))));
 
 	return len_result;
 }
@@ -2805,7 +2881,8 @@ icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar)
  */
 pg_attribute_unused()
 static void
-icu_set_collation_attributes(UCollator *collator, const char *loc)
+icu_set_collation_attributes(pg_icu_library *iculib, UCollator *collator,
+							 const char *loc)
 {
 	char	   *str = asc_tolower(loc, strlen(loc));
 
@@ -2879,7 +2956,7 @@ icu_set_collation_attributes(UCollator *collator, const char *loc)
 				status = U_ILLEGAL_ARGUMENT_ERROR;
 
 			if (status == U_ZERO_ERROR)
-				ucol_setAttribute(collator, uattr, uvalue, &status);
+				iculib->setAttribute(collator, uattr, uvalue, &status);
 
 			/*
 			 * Pretend the error came from ucol_open(), for consistent error
@@ -2888,7 +2965,7 @@ icu_set_collation_attributes(UCollator *collator, const char *loc)
 			if (U_FAILURE(status))
 				ereport(ERROR,
 						(errmsg("could not open collator for locale \"%s\": %s",
-								loc, u_errorName(status))));
+								loc, iculib->errorName(status))));
 		}
 	}
 }
diff --git a/src/include/utils/pg_locale_internal.h b/src/include/utils/pg_locale_internal.h
index 33465ad92d..54445d8b87 100644
--- a/src/include/utils/pg_locale_internal.h
+++ b/src/include/utils/pg_locale_internal.h
@@ -14,6 +14,8 @@
 #define _PG_LOCALE_INTERNAL_
 
 #ifdef USE_ICU
+#include <unicode/ubrk.h>
+#include <unicode/ucnv.h>
 #include <unicode/ucol.h>
 #endif
 
@@ -30,6 +32,104 @@
 #endif
 #endif
 
+#ifdef USE_ICU
+/*
+ * An ICU library version that we're either linked against or have loaded at
+ * runtime.
+ */
+typedef struct pg_icu_library
+{
+	int			major_version;
+	int			minor_version;
+	void		(*getICUVersion) (UVersionInfo info);
+	void		(*getUnicodeVersion) (UVersionInfo into);
+	void		(*getCLDRVersion) (UVersionInfo info, UErrorCode *status);
+	UCollator  *(*openCollator) (const char *loc, UErrorCode *status);
+	void		(*closeCollator) (UCollator *coll);
+	void		(*getCollatorVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*getUCAVersion) (const UCollator *coll, UVersionInfo info);
+	void		(*versionToString) (const UVersionInfo versionArray,
+									char *versionString);
+	UCollationResult (*strcoll) (const UCollator *coll,
+								 const UChar *source,
+								 int32_t sourceLength,
+								 const UChar *target,
+								 int32_t targetLength);
+	UCollationResult (*strcollUTF8) (const UCollator *coll,
+									 const char *source,
+									 int32_t sourceLength,
+									 const char *target,
+									 int32_t targetLength,
+									 UErrorCode *status);
+	int32_t		(*getSortKey) (const UCollator *coll,
+							   const UChar *source,
+							   int32_t sourceLength,
+							   uint8_t *result,
+							   int32_t resultLength);
+	int32_t		(*nextSortKeyPart) (const UCollator *coll,
+									UCharIterator *iter,
+									uint32_t state[2],
+									uint8_t *dest,
+									int32_t count,
+									UErrorCode *status);
+	void		(*setUTF8) (UCharIterator *iter,
+							const char *s,
+							int32_t length);
+	const char *(*errorName) (UErrorCode code);
+	int32_t		(*strToUpper) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToLower) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	int32_t		(*strToTitle) (UChar *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UBreakIterator *titleIter,
+							   const char *locale,
+							   UErrorCode *pErrorCode);
+	void		(*setAttribute) (UCollator *coll,
+								 UColAttribute attr,
+								 UColAttributeValue value,
+								 UErrorCode *status);
+	UConverter *(*openConverter) (const char *converterName,
+								  UErrorCode *  	err);
+	void		(*closeConverter) (UConverter *converter);
+	int32_t		(*fromUChars) (UConverter *cnv,
+							   char *dest,
+							   int32_t destCapacity,
+							   const UChar *src,
+							   int32_t srcLength,
+							   UErrorCode *pErrorCode);
+	int32_t		(*toUChars) (UConverter *cnv,
+							 UChar *dest,
+							 int32_t destCapacity,
+							 const char *src,
+							 int32_t srcLength,
+							 UErrorCode *pErrorCode);
+	int32_t		(*toLanguageTag) (const char *localeID,
+								  char *langtag,
+								  int32_t langtagCapacity,
+								  UBool strict,
+								  UErrorCode *err);
+	int32_t		(*getDisplayName) (const char *localeID,
+								   const char *inLocaleID,
+								   UChar *result,
+								   int32_t maxResultSize,
+								   UErrorCode *err);
+	int32_t		(*countAvailable) (void);
+	const char *(*getAvailable) (int32_t n);
+} pg_icu_library;
+
+#endif
+
 /*
  * We define our own wrapper around locale_t so we can keep the same
  * function signatures for all builds, while not having to create a
@@ -53,7 +153,8 @@ struct pg_locale_struct
 #ifdef USE_ICU
 		struct
 		{
-			UCollator	*ucol;
+			UCollator		*ucol;
+			pg_icu_library	*lib;
 		}			icu;
 #endif
 		int			dummy;		/* in case we have neither LOCALE_T nor ICU */
@@ -61,8 +162,14 @@ struct pg_locale_struct
 };
 
 #ifdef USE_ICU
-extern int32_t icu_to_uchar(UChar **buff_uchar, const char *buff, size_t nbytes);
-extern int32_t icu_from_uchar(char **result, const UChar *buff_uchar, int32_t len_uchar);
+#define PG_ICU_LIB(x) ((x)->info.icu.lib)
+#define PG_ICU_COL(x) ((x)->info.icu.ucol)
+
+extern pg_icu_library *get_builtin_icu_library(void);
+extern int32_t icu_to_uchar(pg_icu_library *lib, UChar **buff_uchar,
+							const char *buff, size_t nbytes);
+extern int32_t icu_from_uchar(pg_icu_library *lib, char **result,
+							  const UChar *buff_uchar, int32_t len_uchar);
 #endif
 
 #endif							/* _PG_LOCALE_INTERNAL_ */
-- 
2.34.1



  [text/x-patch] v2-0005-Add-get_icu_library_hook.patch (6.2K, ../../[email protected]/6-v2-0005-Add-get_icu_library_hook.patch)
  download | inline diff:
From 7dca88aa9f51e7674151bb90506b5305f9107bac Mon Sep 17 00:00:00 2001
From: Jeff Davis <[email protected]>
Date: Mon, 5 Dec 2022 16:14:18 -0800
Subject: [PATCH v2 5/6] Add get_icu_library_hook.

Controls how ICU library symbols are loaded.
---
 src/backend/commands/collationcmds.c   |  6 +--
 src/backend/utils/adt/pg_locale.c      | 56 ++++++++++++++++++++++----
 src/include/utils/pg_locale_internal.h | 10 ++++-
 3 files changed, 61 insertions(+), 11 deletions(-)

diff --git a/src/backend/commands/collationcmds.c b/src/backend/commands/collationcmds.c
index 16b1bcbdc0..f439e832da 100644
--- a/src/backend/commands/collationcmds.c
+++ b/src/backend/commands/collationcmds.c
@@ -560,7 +560,7 @@ get_icu_language_tag(const char *localename)
 {
 	char		buf[ULOC_FULLNAME_CAPACITY];
 	UErrorCode	status;
-	pg_icu_library *iculib = get_builtin_icu_library();
+	pg_icu_library *iculib = get_icu_library(NULL, NULL, NULL);
 
 	status = U_ZERO_ERROR;
 	iculib->toLanguageTag(localename, buf, sizeof(buf), true, &status);
@@ -586,7 +586,7 @@ get_icu_locale_comment(const char *localename)
 	int32		len_uchar;
 	int32		i;
 	char	   *result;
-	pg_icu_library *iculib = get_builtin_icu_library();
+	pg_icu_library *iculib = get_icu_library(NULL, NULL, NULL);
 
 	status = U_ZERO_ERROR;
 	len_uchar = iculib->getDisplayName(localename, "en",
@@ -811,7 +811,7 @@ pg_import_system_collations(PG_FUNCTION_ARGS)
 #ifdef USE_ICU
 	{
 		int			i;
-		pg_icu_library *iculib = get_builtin_icu_library();
+		pg_icu_library *iculib = get_icu_library(NULL, NULL, NULL);
 
 		/*
 		 * Start the loop at -1 to sneak in the root locale without too much
diff --git a/src/backend/utils/adt/pg_locale.c b/src/backend/utils/adt/pg_locale.c
index 4daff8b7b5..b75a825df6 100644
--- a/src/backend/utils/adt/pg_locale.c
+++ b/src/backend/utils/adt/pg_locale.c
@@ -109,6 +109,34 @@ char	   *localized_full_days[7 + 1];
 char	   *localized_abbrev_months[12 + 1];
 char	   *localized_full_months[12 + 1];
 
+/*
+ * get_icu_library_hook can be set to control how the pg_icu_library is
+ * constructed inside a pg_locale_t structure, and therefore which specific
+ * ICU symbols are called.
+ *
+ * Without the hook, Postgres constructs the pg_icu_library from the version
+ * of ICU that Postgres is linked against at build time.
+ *
+ * The hook can instead load the ICU symbols from a different version of the
+ * ICU library on the system, which can avoid problems when the collation
+ * subtly changes across different versions of ICU.
+ *
+ * If the hook returns true, it indicates that it has successfully filled in
+ * the pg_icu_library structure that was passed in. If it returns false,
+ * Postgres will fill in the structure itself. The version of the collation
+ * returned does not need to match exactly the version that was passed in;
+ * though if not, Postgres will issue a WARNING.
+ *
+ * XXX: For now, the only information the hook has access to is the ICU
+ * collation name, ICU ctype, and the ICU version string that was obtained at
+ * the time the collation was created (or when it was last refreshed). We
+ * should consider what other information can be provided to allow for greater
+ * control which library is loaded.
+ */
+#ifdef USE_ICU
+get_icu_library_hook_type get_icu_library_hook = NULL;
+#endif
+
 /* indicates whether locale information cache is valid */
 static bool CurrentLocaleConvValid = false;
 static bool CurrentLCTimeValid = false;
@@ -1463,18 +1491,15 @@ report_newlocale_failure(const char *localename)
 #endif							/* HAVE_LOCALE_T */
 
 #ifdef USE_ICU
-pg_icu_library *
-get_builtin_icu_library()
+static bool
+get_builtin_icu_library(pg_icu_library *lib)
 {
-	pg_icu_library *lib;
-
 	/*
 	 * These assignments will fail to compile if an incompatible API change is
 	 * made to some future version of ICU, at which point we might need to
 	 * consider special treatment for different major version ranges, with
 	 * intermediate trampoline functions.
 	 */
-	lib = palloc0(sizeof(*lib));
 	lib->getICUVersion = u_getVersion;
 	lib->getUnicodeVersion = u_getUnicodeVersion;
 	lib->getCLDRVersion = ulocdata_getCLDRVersion;
@@ -1512,8 +1537,25 @@ get_builtin_icu_library()
 	StaticAssertStmt(U_MAX_VERSION_LENGTH == 4,
 					 "ucol_getVersion output buffer size changed incompatibly");
 
+	return true;
+}
+
+pg_icu_library *
+get_icu_library(const char *collate, const char *ctype, const char *version)
+{
+	pg_icu_library *lib = palloc0(sizeof(*lib));
+	bool filled = false;
+
+	if (get_icu_library_hook != NULL)
+		filled = get_icu_library_hook(lib, collate, ctype, version);
+
+	if(!filled)
+		filled = get_builtin_icu_library(lib);
+
+	Assert(filled);
 	return lib;
 }
+
 #endif
 
 /*
@@ -1607,7 +1649,7 @@ pg_newlocale(char provider, bool isdefault, bool deterministic,
 	{
 		UCollator  *collator;
 		UErrorCode	status;
-		pg_icu_library *iculib = get_builtin_icu_library();
+		pg_icu_library *iculib = get_icu_library(collate, ctype, version);
 
 		/* collator may be leaked if we encounter an error */
 
@@ -1926,7 +1968,7 @@ get_collation_actual_version(char collprovider, const char *collcollate)
 		UErrorCode	status;
 		UVersionInfo versioninfo;
 		char		buf[U_MAX_VERSION_STRING_LENGTH];
-		pg_icu_library	*iculib = get_builtin_icu_library();
+		pg_icu_library	*iculib = get_icu_library(NULL, NULL, NULL);
 
 		status = U_ZERO_ERROR;
 		collator = iculib->openCollator(collcollate, &status);
diff --git a/src/include/utils/pg_locale_internal.h b/src/include/utils/pg_locale_internal.h
index 54445d8b87..1be0205391 100644
--- a/src/include/utils/pg_locale_internal.h
+++ b/src/include/utils/pg_locale_internal.h
@@ -162,10 +162,18 @@ struct pg_locale_struct
 };
 
 #ifdef USE_ICU
+
+typedef bool (*get_icu_library_hook_type)(
+	pg_icu_library *lib, const char *collate, const char *ctype,
+	const char *version);
+
+extern PGDLLIMPORT get_icu_library_hook_type get_icu_library_hook;
+
 #define PG_ICU_LIB(x) ((x)->info.icu.lib)
 #define PG_ICU_COL(x) ((x)->info.icu.ucol)
 
-extern pg_icu_library *get_builtin_icu_library(void);
+extern pg_icu_library *get_icu_library(const char *collate, const char *ctype,
+									   const char *version);
 extern int32_t icu_to_uchar(pg_icu_library *lib, UChar **buff_uchar,
 							const char *buff, size_t nbytes);
 extern int32_t icu_from_uchar(pg_icu_library *lib, char **result,
-- 
2.34.1



^ permalink  raw  reply  [nested|flat] 57+ messages in thread

* [PATCH v20 2/8] Row pattern recognition patch (parse/analysis).
@ 2024-05-24 02:26  Tatsuo Ishii <[email protected]>
  0 siblings, 0 replies; 57+ messages in thread

From: Tatsuo Ishii @ 2024-05-24 02:26 UTC (permalink / raw)

---
 src/backend/parser/parse_agg.c    |   7 +
 src/backend/parser/parse_clause.c | 296 +++++++++++++++++++++++++++++-
 src/backend/parser/parse_expr.c   |   6 +
 src/backend/parser/parse_func.c   |   3 +
 4 files changed, 311 insertions(+), 1 deletion(-)

diff --git a/src/backend/parser/parse_agg.c b/src/backend/parser/parse_agg.c
index bee7d8346a..9bc22a836a 100644
--- a/src/backend/parser/parse_agg.c
+++ b/src/backend/parser/parse_agg.c
@@ -577,6 +577,10 @@ check_agglevels_and_constraints(ParseState *pstate, Node *expr)
 			errkind = true;
 			break;
 
+		case EXPR_KIND_RPR_DEFINE:
+			errkind = true;
+			break;
+
 			/*
 			 * There is intentionally no default: case here, so that the
 			 * compiler will warn if we add a new ParseExprKind without
@@ -967,6 +971,9 @@ transformWindowFuncCall(ParseState *pstate, WindowFunc *wfunc,
 		case EXPR_KIND_CYCLE_MARK:
 			errkind = true;
 			break;
+		case EXPR_KIND_RPR_DEFINE:
+			errkind = true;
+			break;
 
 			/*
 			 * There is intentionally no default: case here, so that the
diff --git a/src/backend/parser/parse_clause.c b/src/backend/parser/parse_clause.c
index 8118036495..9762dce81f 100644
--- a/src/backend/parser/parse_clause.c
+++ b/src/backend/parser/parse_clause.c
@@ -98,7 +98,14 @@ static WindowClause *findWindowClause(List *wclist, const char *name);
 static Node *transformFrameOffset(ParseState *pstate, int frameOptions,
 								  Oid rangeopfamily, Oid rangeopcintype, Oid *inRangeFunc,
 								  Node *clause);
-
+static void transformRPR(ParseState *pstate, WindowClause *wc, WindowDef *windef,
+						 List **targetlist);
+static List *transformDefineClause(ParseState *pstate, WindowClause *wc, WindowDef *windef,
+								   List **targetlist);
+static void transformPatternClause(ParseState *pstate, WindowClause *wc,
+								   WindowDef *windef);
+static List *transformMeasureClause(ParseState *pstate, WindowClause *wc,
+									WindowDef *windef);
 
 /*
  * transformFromClause -
@@ -2956,6 +2963,10 @@ transformWindowDefinitions(ParseState *pstate,
 											 rangeopfamily, rangeopcintype,
 											 &wc->endInRangeFunc,
 											 windef->endOffset);
+
+		/* Process Row Pattern Recognition related clauses */
+		transformRPR(pstate, wc, windef, targetlist);
+
 		wc->winref = winref;
 
 		result = lappend(result, wc);
@@ -3820,3 +3831,286 @@ transformFrameOffset(ParseState *pstate, int frameOptions,
 
 	return node;
 }
+
+/*
+ * transformRPR
+ *		Process Row Pattern Recognition related clauses
+ */
+static void
+transformRPR(ParseState *pstate, WindowClause *wc, WindowDef *windef,
+			 List **targetlist)
+{
+	/*
+	 * Window definition exists?
+	 */
+	if (windef == NULL)
+		return;
+
+	/*
+	 * Row Pattern Common Syntax clause exists?
+	 */
+	if (windef->rpCommonSyntax == NULL)
+		return;
+
+	/* Check Frame option. Frame must start at current row */
+	if ((wc->frameOptions & FRAMEOPTION_START_CURRENT_ROW) == 0)
+		ereport(ERROR,
+				(errcode(ERRCODE_SYNTAX_ERROR),
+				 errmsg("FRAME must start at current row when row patttern recognition is used")));
+
+	/* Transform AFTER MACH SKIP TO clause */
+	wc->rpSkipTo = windef->rpCommonSyntax->rpSkipTo;
+
+	/* Transform AFTER MACH SKIP TO variable */
+	wc->rpSkipVariable = windef->rpCommonSyntax->rpSkipVariable;
+
+	/* Transform SEEK or INITIAL clause */
+	wc->initial = windef->rpCommonSyntax->initial;
+
+	/* Transform DEFINE clause into list of TargetEntry's */
+	wc->defineClause = transformDefineClause(pstate, wc, windef, targetlist);
+
+	/* Check PATTERN clause and copy to patternClause */
+	transformPatternClause(pstate, wc, windef);
+
+	/* Transform MEASURE clause */
+	transformMeasureClause(pstate, wc, windef);
+}
+
+/*
+ * transformDefineClause Process DEFINE clause and transform ResTarget into
+ *		list of TargetEntry.
+ *
+ * XXX we only support column reference in row pattern definition search
+ * condition, e.g. "price". <row pattern definition variable name>.<column
+ * reference> is not supported, e.g. "A.price".
+ */
+static List *
+transformDefineClause(ParseState *pstate, WindowClause *wc, WindowDef *windef,
+					  List **targetlist)
+{
+	/* DEFINE variable name initials */
+	static char *defineVariableInitials = "abcdefghijklmnopqrstuvwxyz";
+
+	ListCell   *lc,
+			   *l;
+	ResTarget  *restarget,
+			   *r;
+	List	   *restargets;
+	List	   *defineClause;
+	char	   *name;
+	int			initialLen;
+	int			i;
+
+	/*
+	 * If Row Definition Common Syntax exists, DEFINE clause must exist. (the
+	 * raw parser should have already checked it.)
+	 */
+	Assert(windef->rpCommonSyntax->rpDefs != NULL);
+
+	/*
+	 * Check and add "A AS A IS TRUE" if pattern variable is missing in DEFINE
+	 * per the SQL standard.
+	 */
+	restargets = NIL;
+	foreach(lc, windef->rpCommonSyntax->rpPatterns)
+	{
+		A_Expr	   *a;
+		bool		found = false;
+
+		if (!IsA(lfirst(lc), A_Expr))
+			ereport(ERROR,
+					errmsg("node type is not A_Expr"));
+
+		a = (A_Expr *) lfirst(lc);
+		name = strVal(a->lexpr);
+
+		foreach(l, windef->rpCommonSyntax->rpDefs)
+		{
+			restarget = (ResTarget *) lfirst(l);
+
+			if (!strcmp(restarget->name, name))
+			{
+				found = true;
+				break;
+			}
+		}
+
+		if (!found)
+		{
+			/*
+			 * "name" is missing. So create "name AS name IS TRUE" ResTarget
+			 * node and add it to the temporary list.
+			 */
+			A_Const    *n;
+
+			restarget = makeNode(ResTarget);
+			n = makeNode(A_Const);
+			n->val.boolval.type = T_Boolean;
+			n->val.boolval.boolval = true;
+			n->location = -1;
+			restarget->name = pstrdup(name);
+			restarget->indirection = NIL;
+			restarget->val = (Node *) n;
+			restarget->location = -1;
+			restargets = lappend((List *) restargets, restarget);
+		}
+	}
+
+	if (list_length(restargets) >= 1)
+	{
+		/* add missing DEFINEs */
+		windef->rpCommonSyntax->rpDefs =
+			list_concat(windef->rpCommonSyntax->rpDefs, restargets);
+		list_free(restargets);
+	}
+
+	/*
+	 * Check for duplicate row pattern definition variables.  The standard
+	 * requires that no two row pattern definition variable names shall be
+	 * equivalent.
+	 */
+	restargets = NIL;
+	foreach(lc, windef->rpCommonSyntax->rpDefs)
+	{
+		restarget = (ResTarget *) lfirst(lc);
+		name = restarget->name;
+
+		/*
+		 * Add DEFINE expression (Restarget->val) to the targetlist as a
+		 * TargetEntry if it does not exist yet. Planner will add the column
+		 * ref var node to the outer plan's target list later on. This makes
+		 * DEFINE expression could access the outer tuple while evaluating
+		 * PATTERN.
+		 *
+		 * XXX: adding whole expressions of DEFINE to the plan.targetlist is
+		 * not so good, because it's not necessary to evalute the expression
+		 * in the target list while running the plan. We should extract the
+		 * var nodes only then add them to the plan.targetlist.
+		 */
+		findTargetlistEntrySQL99(pstate, (Node *) restarget->val,
+								 targetlist, EXPR_KIND_RPR_DEFINE);
+
+		/*
+		 * Make sure that the row pattern definition search condition is a
+		 * boolean expression.
+		 */
+		transformWhereClause(pstate, restarget->val,
+							 EXPR_KIND_RPR_DEFINE, "DEFINE");
+
+		foreach(l, restargets)
+		{
+			char	   *n;
+
+			r = (ResTarget *) lfirst(l);
+			n = r->name;
+
+			if (!strcmp(n, name))
+				ereport(ERROR,
+						(errcode(ERRCODE_SYNTAX_ERROR),
+						 errmsg("row pattern definition variable name \"%s\" appears more than once in DEFINE clause",
+								name),
+						 parser_errposition(pstate, exprLocation((Node *) r))));
+		}
+		restargets = lappend(restargets, restarget);
+	}
+	list_free(restargets);
+
+	/*
+	 * Create list of row pattern DEFINE variable name's initial. We assign
+	 * [a-z] to them (up to 26 variable names are allowed).
+	 */
+	restargets = NIL;
+	i = 0;
+	initialLen = strlen(defineVariableInitials);
+
+	foreach(lc, windef->rpCommonSyntax->rpDefs)
+	{
+		char		initial[2];
+
+		restarget = (ResTarget *) lfirst(lc);
+		name = restarget->name;
+
+		if (i >= initialLen)
+		{
+			ereport(ERROR,
+					(errcode(ERRCODE_SYNTAX_ERROR),
+					 errmsg("number of row pattern definition variable names exceeds %d",
+							initialLen),
+					 parser_errposition(pstate,
+										exprLocation((Node *) restarget))));
+		}
+		initial[0] = defineVariableInitials[i++];
+		initial[1] = '\0';
+		wc->defineInitial = lappend(wc->defineInitial,
+									makeString(pstrdup(initial)));
+	}
+
+	defineClause = transformTargetList(pstate, windef->rpCommonSyntax->rpDefs,
+									   EXPR_KIND_RPR_DEFINE);
+
+	/* mark column origins */
+	markTargetListOrigins(pstate, defineClause);
+
+	/* mark all nodes in the DEFINE clause tree with collation information */
+	assign_expr_collations(pstate, (Node *) defineClause);
+
+	return defineClause;
+}
+
+/*
+ * transformPatternClause
+ *		Process PATTERN clause and return PATTERN clause in the raw parse tree
+ */
+static void
+transformPatternClause(ParseState *pstate, WindowClause *wc,
+					   WindowDef *windef)
+{
+	ListCell   *lc;
+
+	/*
+	 * Row Pattern Common Syntax clause exists?
+	 */
+	if (windef->rpCommonSyntax == NULL)
+		return;
+
+	wc->patternVariable = NIL;
+	wc->patternRegexp = NIL;
+	foreach(lc, windef->rpCommonSyntax->rpPatterns)
+	{
+		A_Expr	   *a;
+		char	   *name;
+		char	   *regexp;
+
+		if (!IsA(lfirst(lc), A_Expr))
+			ereport(ERROR,
+					errmsg("node type is not A_Expr"));
+
+		a = (A_Expr *) lfirst(lc);
+		name = strVal(a->lexpr);
+
+		wc->patternVariable = lappend(wc->patternVariable, makeString(pstrdup(name)));
+		regexp = strVal(lfirst(list_head(a->name)));
+
+		wc->patternRegexp = lappend(wc->patternRegexp, makeString(pstrdup(regexp)));
+	}
+}
+
+/*
+ * transformMeasureClause
+ *		Process MEASURE clause
+ *	XXX MEASURE clause is not supported yet
+ */
+static List *
+transformMeasureClause(ParseState *pstate, WindowClause *wc,
+					   WindowDef *windef)
+{
+	if (windef->rowPatternMeasures == NIL)
+		return NIL;
+
+	ereport(ERROR,
+			(errcode(ERRCODE_SYNTAX_ERROR),
+			 errmsg("%s", "MEASURE clause is not supported yet"),
+			 parser_errposition(pstate, exprLocation((Node *) windef->rowPatternMeasures))));
+	return NIL;
+}
diff --git a/src/backend/parser/parse_expr.c b/src/backend/parser/parse_expr.c
index aba3546ed1..e98b45e06e 100644
--- a/src/backend/parser/parse_expr.c
+++ b/src/backend/parser/parse_expr.c
@@ -578,6 +578,7 @@ transformColumnRef(ParseState *pstate, ColumnRef *cref)
 		case EXPR_KIND_COPY_WHERE:
 		case EXPR_KIND_GENERATED_COLUMN:
 		case EXPR_KIND_CYCLE_MARK:
+		case EXPR_KIND_RPR_DEFINE:
 			/* okay */
 			break;
 
@@ -1860,6 +1861,9 @@ transformSubLink(ParseState *pstate, SubLink *sublink)
 		case EXPR_KIND_GENERATED_COLUMN:
 			err = _("cannot use subquery in column generation expression");
 			break;
+		case EXPR_KIND_RPR_DEFINE:
+			err = _("cannot use subquery in DEFINE expression");
+			break;
 
 			/*
 			 * There is intentionally no default: case here, so that the
@@ -3197,6 +3201,8 @@ ParseExprKindName(ParseExprKind exprKind)
 			return "GENERATED AS";
 		case EXPR_KIND_CYCLE_MARK:
 			return "CYCLE";
+		case EXPR_KIND_RPR_DEFINE:
+			return "DEFINE";
 
 			/*
 			 * There is intentionally no default: case here, so that the
diff --git a/src/backend/parser/parse_func.c b/src/backend/parser/parse_func.c
index 9b23344a3b..4c482abb30 100644
--- a/src/backend/parser/parse_func.c
+++ b/src/backend/parser/parse_func.c
@@ -2658,6 +2658,9 @@ check_srf_call_placement(ParseState *pstate, Node *last_srf, int location)
 		case EXPR_KIND_CYCLE_MARK:
 			errkind = true;
 			break;
+		case EXPR_KIND_RPR_DEFINE:
+			errkind = true;
+			break;
 
 			/*
 			 * There is intentionally no default: case here, so that the
-- 
2.25.1


----Next_Part(Fri_May_24_11_39_19_2024_763)--
Content-Type: Text/X-Patch; charset=us-ascii
Content-Transfer-Encoding: 7bit
Content-Disposition: inline;
 filename="v20-0003-Row-pattern-recognition-patch-rewriter.patch"



^ permalink  raw  reply  [nested|flat] 57+ messages in thread


end of thread, other threads:[~2024-05-24 02:26 UTC | newest]

Thread overview: 57+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2021-01-22 00:11 [PATCH 8/8] batch build Tomas Vondra <[email protected]>
2022-10-21 21:24 Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-10-22 01:22 ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-01 10:33   ` Re: Collation version tracking for macOS Peter Eisentraut <[email protected]>
2022-11-01 12:42     ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-01 23:57       ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-07 12:21         ` Re: Collation version tracking for macOS Peter Eisentraut <[email protected]>
2022-11-09 02:37           ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-11 14:57   ` Re: Collation version tracking for macOS Peter Eisentraut <[email protected]>
2022-11-15 00:55   ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-18 18:38     ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-18 19:16       ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-22 06:34       ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-23 05:08         ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-24 02:07           ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-24 02:57             ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-24 04:48             ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-26 05:27               ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-28 06:10                 ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 02:54                 ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 02:57                   ` Re: Collation version tracking for macOS Robert Haas <[email protected]>
2022-11-29 03:36                     ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 19:37                       ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-12-01 13:22                         ` Re: Collation version tracking for macOS Dagfinn Ilmari Mannsåker <[email protected]>
2022-11-29 04:34                   ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 18:03                   ` Re: Collation version tracking for macOS Jeremy Schneider <[email protected]>
2022-11-29 18:18                     ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 19:03                       ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 19:41                         ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 20:59                           ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 21:29                             ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-30 00:32                               ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-30 00:50                                 ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-30 01:00                                   ` Re: Collation version tracking for macOS Michael Paquier <[email protected]>
2022-11-29 06:51                 ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-12-05 03:12                   ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-12-05 15:45                     ` Re: Collation version tracking for macOS Robert Haas <[email protected]>
2022-12-05 17:41                     ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-12-05 17:45                       ` Re: Collation version tracking for macOS Joe Conway <[email protected]>
2022-12-05 21:33                         ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-12-08 05:56                           ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-28 19:11           ` Re: Collation version tracking for macOS Robert Haas <[email protected]>
2022-11-29 04:48             ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 17:32               ` Re: Collation version tracking for macOS Robert Haas <[email protected]>
2022-11-29 18:46                 ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 19:38                   ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 21:52                     ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-30 00:25                       ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-30 00:54                         ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 16:27             ` Re: Collation version tracking for macOS Joe Conway <[email protected]>
2022-11-29 18:59               ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 19:34                 ` Re: Collation version tracking for macOS Joe Conway <[email protected]>
2022-11-29 20:21                   ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2022-11-29 19:52                 ` Re: Collation version tracking for macOS Robert Haas <[email protected]>
2022-11-29 20:00                   ` Re: Collation version tracking for macOS Thomas Munro <[email protected]>
2022-11-29 21:41                     ` Re: Collation version tracking for macOS Jeff Davis <[email protected]>
2024-05-24 02:26 [PATCH v20 2/8] Row pattern recognition patch (parse/analysis). Tatsuo Ishii <[email protected]>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox